{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:2G6OBEHEFEDVGQSOZS44RUKJNT","short_pith_number":"pith:2G6OBEHE","schema_version":"1.0","canonical_sha256":"d1bce090e4290753424eccb9c8d1496cd7353204b44d6b253eaa40d5d9bd01ad","source":{"kind":"arxiv","id":"2402.17139","version":1},"attestation_state":"computed","paper":{"title":"Video as the New Language for Real-World Decision Making","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Andre Barreto, Dale Schuurmans, Jack Parker-Holder, Jacob Walker, Jake Bruce, Pieter Abbeel, Sherry Yang, Yilun Du","submitted_at":"2024-02-27T02:05:29Z","abstract_excerpt":"Both text and video data are abundant on the internet and support large-scale self-supervised learning through next token or frame prediction. However, they have not been equally leveraged: language models have had significant real-world impact, whereas video generation has remained largely limited to media entertainment. Yet video data captures important information about the physical world that is difficult to express in language. To address this gap, we discuss an under-appreciated opportunity to extend video generation to solve tasks in the real world. We observe how, akin to language, vid"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2402.17139","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2024-02-27T02:05:29Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"317e520322ed8c56b339f6145d22424d5412c0c5d99c0c19a1badfc8e2daf012","abstract_canon_sha256":"b11b392e593289112ece8de21680386783ddee90e3ad7fc4a21839a422567ff3"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:49:43.108603Z","signature_b64":"3Zv5hMrP10hvLQUhg3y7qTyvUoEMkBCCDXLGwD8IJoM/Egzfh/HIVtuyCiqW2/i9XFhxV1/gjHAS602yU+KpDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"d1bce090e4290753424eccb9c8d1496cd7353204b44d6b253eaa40d5d9bd01ad","last_reissued_at":"2026-07-05T07:49:43.108107Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:49:43.108107Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Video as the New Language for Real-World Decision Making","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Andre Barreto, Dale Schuurmans, Jack Parker-Holder, Jacob Walker, Jake Bruce, Pieter Abbeel, Sherry Yang, Yilun Du","submitted_at":"2024-02-27T02:05:29Z","abstract_excerpt":"Both text and video data are abundant on the internet and support large-scale self-supervised learning through next token or frame prediction. However, they have not been equally leveraged: language models have had significant real-world impact, whereas video generation has remained largely limited to media entertainment. Yet video data captures important information about the physical world that is difficult to express in language. To address this gap, we discuss an under-appreciated opportunity to extend video generation to solve tasks in the real world. We observe how, akin to language, vid"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2402.17139","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2402.17139/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2402.17139","created_at":"2026-07-05T07:49:43.108162+00:00"},{"alias_kind":"arxiv_version","alias_value":"2402.17139v1","created_at":"2026-07-05T07:49:43.108162+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2402.17139","created_at":"2026-07-05T07:49:43.108162+00:00"},{"alias_kind":"pith_short_12","alias_value":"2G6OBEHEFEDV","created_at":"2026-07-05T07:49:43.108162+00:00"},{"alias_kind":"pith_short_16","alias_value":"2G6OBEHEFEDVGQSO","created_at":"2026-07-05T07:49:43.108162+00:00"},{"alias_kind":"pith_short_8","alias_value":"2G6OBEHE","created_at":"2026-07-05T07:49:43.108162+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.01896","citing_title":"Divide and Conquer: Decoupled Representation Alignment for Multimodal World Models","ref_index":53,"is_internal_anchor":false},{"citing_arxiv_id":"2512.00961","citing_title":"Goal-Driven Reward by Video Diffusion Models for Reinforcement Learning","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2503.22020","citing_title":"CoT-VLA: Visual Chain-of-Thought Reasoning for Vision-Language-Action Models","ref_index":73,"is_internal_anchor":false},{"citing_arxiv_id":"2409.16283","citing_title":"Gen2Act: Human Video Generation in Novel Scenarios enables Generalizable Robot Manipulation","ref_index":36,"is_internal_anchor":false},{"citing_arxiv_id":"2509.20328","citing_title":"Video models are zero-shot learners and reasoners","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01896","citing_title":"Divide and Conquer: Decoupled Representation Alignment for Multimodal World Models","ref_index":53,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/2G6OBEHEFEDVGQSOZS44RUKJNT","json":"https://pith.science/pith/2G6OBEHEFEDVGQSOZS44RUKJNT.json","graph_json":"https://pith.science/api/pith-number/2G6OBEHEFEDVGQSOZS44RUKJNT/graph.json","events_json":"https://pith.science/api/pith-number/2G6OBEHEFEDVGQSOZS44RUKJNT/events.json","paper":"https://pith.science/paper/2G6OBEHE"},"agent_actions":{"view_html":"https://pith.science/pith/2G6OBEHEFEDVGQSOZS44RUKJNT","download_json":"https://pith.science/pith/2G6OBEHEFEDVGQSOZS44RUKJNT.json","view_paper":"https://pith.science/paper/2G6OBEHE","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2402.17139&json=true","fetch_graph":"https://pith.science/api/pith-number/2G6OBEHEFEDVGQSOZS44RUKJNT/graph.json","fetch_events":"https://pith.science/api/pith-number/2G6OBEHEFEDVGQSOZS44RUKJNT/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/2G6OBEHEFEDVGQSOZS44RUKJNT/action/timestamp_anchor","attest_storage":"https://pith.science/pith/2G6OBEHEFEDVGQSOZS44RUKJNT/action/storage_attestation","attest_author":"https://pith.science/pith/2G6OBEHEFEDVGQSOZS44RUKJNT/action/author_attestation","sign_citation":"https://pith.science/pith/2G6OBEHEFEDVGQSOZS44RUKJNT/action/citation_signature","submit_replication":"https://pith.science/pith/2G6OBEHEFEDVGQSOZS44RUKJNT/action/replication_record"}},"created_at":"2026-07-05T07:49:43.108162+00:00","updated_at":"2026-07-05T07:49:43.108162+00:00"}