{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2019:HVRD2KLN7WXYEC3SVHADQXRVD4","short_pith_number":"pith:HVRD2KLN","schema_version":"1.0","canonical_sha256":"3d623d296dfdaf820b72a9c0385e351f3ca053147ef47bd414ec00e3abab5c22","source":{"kind":"arxiv","id":"1909.12200","version":3},"attestation_state":"computed","paper":{"title":"Scaling data-driven robotics with reward sketching and batch reinforcement learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.RO","authors_text":"Alexander Novikov, David Barker, David Budden, Jonathan Scholz, Konrad Zolna, Ksenia Konyushkova, Mel Vecerik, Misha Denil, Nando de Freitas, Oleg Sushkov, Rae Jeong, Scott Reed, Sergio G\\'omez Colmenarejo, Serkan Cabi, Yusuf Aytar, Ziyu Wang","submitted_at":"2019-09-26T15:45:23Z","abstract_excerpt":"We present a framework for data-driven robotics that makes use of a large dataset of recorded robot experience and scales to several tasks using learned reward functions. We show how to apply this framework to accomplish three different object manipulation tasks on a real robot platform. Given demonstrations of a task together with task-agnostic recorded experience, we use a special form of human annotation as supervision to learn a reward function, which enables us to deal with real-world tasks where the reward signal cannot be acquired directly. Learned rewards are used in combination with a"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"1909.12200","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.RO","submitted_at":"2019-09-26T15:45:23Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"19d7e1baf743c631667ab2a807c929b3300ec76b88867fd2aac05f3662192614","abstract_canon_sha256":"a57c54dbe0d832f606b34fb00fb05932ab4a8acc4b2225494a60d6ca36349369"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T01:07:50.673942Z","signature_b64":"pr9ACDzqr7UK320CWX2SGmkFAfQbkmX9+tefzalzk3Fcu7byDzG0xp9TUP+7N85wdmL80Cz+KKOFf7/QOLMBBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"3d623d296dfdaf820b72a9c0385e351f3ca053147ef47bd414ec00e3abab5c22","last_reissued_at":"2026-07-05T01:07:50.673438Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T01:07:50.673438Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Scaling data-driven robotics with reward sketching and batch reinforcement learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.RO","authors_text":"Alexander Novikov, David Barker, David Budden, Jonathan Scholz, Konrad Zolna, Ksenia Konyushkova, Mel Vecerik, Misha Denil, Nando de Freitas, Oleg Sushkov, Rae Jeong, Scott Reed, Sergio G\\'omez Colmenarejo, Serkan Cabi, Yusuf Aytar, Ziyu Wang","submitted_at":"2019-09-26T15:45:23Z","abstract_excerpt":"We present a framework for data-driven robotics that makes use of a large dataset of recorded robot experience and scales to several tasks using learned reward functions. We show how to apply this framework to accomplish three different object manipulation tasks on a real robot platform. Given demonstrations of a task together with task-agnostic recorded experience, we use a special form of human annotation as supervision to learn a reward function, which enables us to deal with real-world tasks where the reward signal cannot be acquired directly. Learned rewards are used in combination with a"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1909.12200","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/1909.12200/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"1909.12200","created_at":"2026-07-05T01:07:50.673496+00:00"},{"alias_kind":"arxiv_version","alias_value":"1909.12200v3","created_at":"2026-07-05T01:07:50.673496+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1909.12200","created_at":"2026-07-05T01:07:50.673496+00:00"},{"alias_kind":"pith_short_12","alias_value":"HVRD2KLN7WXY","created_at":"2026-07-05T01:07:50.673496+00:00"},{"alias_kind":"pith_short_16","alias_value":"HVRD2KLN7WXYEC3S","created_at":"2026-07-05T01:07:50.673496+00:00"},{"alias_kind":"pith_short_8","alias_value":"HVRD2KLN","created_at":"2026-07-05T01:07:50.673496+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2205.06175","citing_title":"A Generalist Agent","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2004.07219","citing_title":"D4RL: Datasets for Deep Data-Driven Reinforcement Learning","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2005.01643","citing_title":"Offline Reinforcement Learning: Tutorial, Review, and Perspectives on Open Problems","ref_index":279,"is_internal_anchor":false},{"citing_arxiv_id":"2405.12213","citing_title":"Octo: An Open-Source Generalist Robot Policy","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2410.24164","citing_title":"$\\pi_0$: A Vision-Language-Action Flow Model for General Robot Control","ref_index":8,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/HVRD2KLN7WXYEC3SVHADQXRVD4","json":"https://pith.science/pith/HVRD2KLN7WXYEC3SVHADQXRVD4.json","graph_json":"https://pith.science/api/pith-number/HVRD2KLN7WXYEC3SVHADQXRVD4/graph.json","events_json":"https://pith.science/api/pith-number/HVRD2KLN7WXYEC3SVHADQXRVD4/events.json","paper":"https://pith.science/paper/HVRD2KLN"},"agent_actions":{"view_html":"https://pith.science/pith/HVRD2KLN7WXYEC3SVHADQXRVD4","download_json":"https://pith.science/pith/HVRD2KLN7WXYEC3SVHADQXRVD4.json","view_paper":"https://pith.science/paper/HVRD2KLN","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=1909.12200&json=true","fetch_graph":"https://pith.science/api/pith-number/HVRD2KLN7WXYEC3SVHADQXRVD4/graph.json","fetch_events":"https://pith.science/api/pith-number/HVRD2KLN7WXYEC3SVHADQXRVD4/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/HVRD2KLN7WXYEC3SVHADQXRVD4/action/timestamp_anchor","attest_storage":"https://pith.science/pith/HVRD2KLN7WXYEC3SVHADQXRVD4/action/storage_attestation","attest_author":"https://pith.science/pith/HVRD2KLN7WXYEC3SVHADQXRVD4/action/author_attestation","sign_citation":"https://pith.science/pith/HVRD2KLN7WXYEC3SVHADQXRVD4/action/citation_signature","submit_replication":"https://pith.science/pith/HVRD2KLN7WXYEC3SVHADQXRVD4/action/replication_record"}},"created_at":"2026-07-05T01:07:50.673496+00:00","updated_at":"2026-07-05T01:07:50.673496+00:00"}