{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:74PL5O3MKIMMPDMZKVBPFYENN4","short_pith_number":"pith:74PL5O3M","schema_version":"1.0","canonical_sha256":"ff1ebebb6c5218c78d995542f2e08d6f3b2e3c267a0f96e08ae68f2e79cb7213","source":{"kind":"arxiv","id":"2309.13041","version":1},"attestation_state":"computed","paper":{"title":"Robotic Offline RL from Internet Videos via Value-Function Pre-Training","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CV","cs.LG"],"primary_cat":"cs.RO","authors_text":"Anikait Singh, Aviral Kumar, Chethan Bhateja, Derek Guo, Dibya Ghosh, Manan Tomar, Quan Vuong, Sergey Levine, Yevgen Chebotar","submitted_at":"2023-09-22T17:59:14Z","abstract_excerpt":"Pre-training on Internet data has proven to be a key ingredient for broad generalization in many modern ML systems. What would it take to enable such capabilities in robotic reinforcement learning (RL)? Offline RL methods, which learn from datasets of robot experience, offer one way to leverage prior data into the robotic learning pipeline. However, these methods have a \"type mismatch\" with video data (such as Ego4D), the largest prior datasets available for robotics, since video offers observation-only experience without the action or reward annotations needed for RL methods. In this paper, w"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2309.13041","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.RO","submitted_at":"2023-09-22T17:59:14Z","cross_cats_sorted":["cs.CV","cs.LG"],"title_canon_sha256":"7b6ab7048aa958e539fb0d859d54bbc1a5fbd538b79772e004f6849bd11a49a5","abstract_canon_sha256":"cd5e4e6cb5a0bd47833ffec51d2a8268c1ee8990498607c6604fc3ddfb9deb0f"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:53:26.623431Z","signature_b64":"JW3B2AAwjoBk+j0i4RVxG7YXopVeYauDz6hBpKgAX1Ctpzs5Kv3MOfd0IxxxNJd8nf9BlzeIOkcMcZUQL3epDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ff1ebebb6c5218c78d995542f2e08d6f3b2e3c267a0f96e08ae68f2e79cb7213","last_reissued_at":"2026-07-05T06:53:26.622942Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:53:26.622942Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Robotic Offline RL from Internet Videos via Value-Function Pre-Training","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CV","cs.LG"],"primary_cat":"cs.RO","authors_text":"Anikait Singh, Aviral Kumar, Chethan Bhateja, Derek Guo, Dibya Ghosh, Manan Tomar, Quan Vuong, Sergey Levine, Yevgen Chebotar","submitted_at":"2023-09-22T17:59:14Z","abstract_excerpt":"Pre-training on Internet data has proven to be a key ingredient for broad generalization in many modern ML systems. What would it take to enable such capabilities in robotic reinforcement learning (RL)? Offline RL methods, which learn from datasets of robot experience, offer one way to leverage prior data into the robotic learning pipeline. However, these methods have a \"type mismatch\" with video data (such as Ego4D), the largest prior datasets available for robotics, since video offers observation-only experience without the action or reward annotations needed for RL methods. In this paper, w"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2309.13041","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2309.13041/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2309.13041","created_at":"2026-07-05T06:53:26.622998+00:00"},{"alias_kind":"arxiv_version","alias_value":"2309.13041v1","created_at":"2026-07-05T06:53:26.622998+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2309.13041","created_at":"2026-07-05T06:53:26.622998+00:00"},{"alias_kind":"pith_short_12","alias_value":"74PL5O3MKIMM","created_at":"2026-07-05T06:53:26.622998+00:00"},{"alias_kind":"pith_short_16","alias_value":"74PL5O3MKIMMPDMZ","created_at":"2026-07-05T06:53:26.622998+00:00"},{"alias_kind":"pith_short_8","alias_value":"74PL5O3M","created_at":"2026-07-05T06:53:26.622998+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2312.13139","citing_title":"Unleashing Large-Scale Video Generative Pre-training for Visual Robot Manipulation","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11832","citing_title":"Learning Action Manifold with Multi-view Latent Priors for Robotic Manipulation","ref_index":44,"is_internal_anchor":false},{"citing_arxiv_id":"2604.15483","citing_title":"${\\pi}_{0.7}$: a Steerable Generalist Robotic Foundation Model with Emergent Capabilities","ref_index":58,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/74PL5O3MKIMMPDMZKVBPFYENN4","json":"https://pith.science/pith/74PL5O3MKIMMPDMZKVBPFYENN4.json","graph_json":"https://pith.science/api/pith-number/74PL5O3MKIMMPDMZKVBPFYENN4/graph.json","events_json":"https://pith.science/api/pith-number/74PL5O3MKIMMPDMZKVBPFYENN4/events.json","paper":"https://pith.science/paper/74PL5O3M"},"agent_actions":{"view_html":"https://pith.science/pith/74PL5O3MKIMMPDMZKVBPFYENN4","download_json":"https://pith.science/pith/74PL5O3MKIMMPDMZKVBPFYENN4.json","view_paper":"https://pith.science/paper/74PL5O3M","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2309.13041&json=true","fetch_graph":"https://pith.science/api/pith-number/74PL5O3MKIMMPDMZKVBPFYENN4/graph.json","fetch_events":"https://pith.science/api/pith-number/74PL5O3MKIMMPDMZKVBPFYENN4/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/74PL5O3MKIMMPDMZKVBPFYENN4/action/timestamp_anchor","attest_storage":"https://pith.science/pith/74PL5O3MKIMMPDMZKVBPFYENN4/action/storage_attestation","attest_author":"https://pith.science/pith/74PL5O3MKIMMPDMZKVBPFYENN4/action/author_attestation","sign_citation":"https://pith.science/pith/74PL5O3MKIMMPDMZKVBPFYENN4/action/citation_signature","submit_replication":"https://pith.science/pith/74PL5O3MKIMMPDMZKVBPFYENN4/action/replication_record"}},"created_at":"2026-07-05T06:53:26.622998+00:00","updated_at":"2026-07-05T06:53:26.622998+00:00"}