{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:OJHYRWXLFEBEWU7C7BAAL727BM","short_pith_number":"pith:OJHYRWXL","schema_version":"1.0","canonical_sha256":"724f88daeb29024b53e2f84005ff5f0b1c6ee0fd7d4308153881dcece100fa32","source":{"kind":"arxiv","id":"2410.07169","version":2},"attestation_state":"computed","paper":{"title":"VIP: Vision Instructed Pre-training for Robotic Manipulation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.RO","authors_text":"Hengshuang Zhao, Jinrong Yang, Liangliang Ren, Xiang Bai, Xiaoyang Wu, Yong Zhao, Zhenhua Xu, Zhuoling Li","submitted_at":"2024-10-09T17:59:06Z","abstract_excerpt":"The effectiveness of scaling up training data in robotic manipulation is still limited. A primary challenge in manipulation is the tasks are diverse, and the trained policy would be confused if the task targets are not specified clearly. Existing works primarily rely on text instruction to describe targets. However, we reveal that current robotic data cannot train policies to understand text instruction effectively, and vision is much more comprehensible. Therefore, we introduce utilizing vision instruction to specify targets. A straightforward implementation is training a policy to predict th"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.07169","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.RO","submitted_at":"2024-10-09T17:59:06Z","cross_cats_sorted":[],"title_canon_sha256":"0e60240cb8b9b8f63daee35449ba47b4a49395794156db7ebb43ac1fbe0e21b9","abstract_canon_sha256":"e9fe8b8e356f8db36a798c60f3ae04b6573f75cb36bf5ccd37562b03c33afa49"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:12:49.412687Z","signature_b64":"PHPxl/YuLFoDVsEzw3HY2qP9qQVhK+W17SHpEsPCr3sUcXiIPWaRwBl3eQggemSdG2meei/CrGfH295aOuC7CQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"724f88daeb29024b53e2f84005ff5f0b1c6ee0fd7d4308153881dcece100fa32","last_reissued_at":"2026-07-05T10:12:49.412078Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:12:49.412078Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"VIP: Vision Instructed Pre-training for Robotic Manipulation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.RO","authors_text":"Hengshuang Zhao, Jinrong Yang, Liangliang Ren, Xiang Bai, Xiaoyang Wu, Yong Zhao, Zhenhua Xu, Zhuoling Li","submitted_at":"2024-10-09T17:59:06Z","abstract_excerpt":"The effectiveness of scaling up training data in robotic manipulation is still limited. A primary challenge in manipulation is the tasks are diverse, and the trained policy would be confused if the task targets are not specified clearly. Existing works primarily rely on text instruction to describe targets. However, we reveal that current robotic data cannot train policies to understand text instruction effectively, and vision is much more comprehensible. Therefore, we introduce utilizing vision instruction to specify targets. A straightforward implementation is training a policy to predict th"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.07169","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.07169/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.07169","created_at":"2026-07-05T10:12:49.412170+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.07169v2","created_at":"2026-07-05T10:12:49.412170+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.07169","created_at":"2026-07-05T10:12:49.412170+00:00"},{"alias_kind":"pith_short_12","alias_value":"OJHYRWXLFEBE","created_at":"2026-07-05T10:12:49.412170+00:00"},{"alias_kind":"pith_short_16","alias_value":"OJHYRWXLFEBEWU7C","created_at":"2026-07-05T10:12:49.412170+00:00"},{"alias_kind":"pith_short_8","alias_value":"OJHYRWXL","created_at":"2026-07-05T10:12:49.412170+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.06564","citing_title":"Lift3D-VLA: Lifting VLA Models to 3D Geometry and Dynamics-Aware Manipulation","ref_index":39,"is_internal_anchor":true},{"citing_arxiv_id":"2603.23202","citing_title":"Gaze-Regularized Vision-Language-Action Models for Robotic Manipulation","ref_index":55,"is_internal_anchor":false},{"citing_arxiv_id":"2604.10432","citing_title":"AnySlot: Goal-Conditioned Vision-Language-Action Policies for Zero-Shot Slot-Level Placement","ref_index":18,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/OJHYRWXLFEBEWU7C7BAAL727BM","json":"https://pith.science/pith/OJHYRWXLFEBEWU7C7BAAL727BM.json","graph_json":"https://pith.science/api/pith-number/OJHYRWXLFEBEWU7C7BAAL727BM/graph.json","events_json":"https://pith.science/api/pith-number/OJHYRWXLFEBEWU7C7BAAL727BM/events.json","paper":"https://pith.science/paper/OJHYRWXL"},"agent_actions":{"view_html":"https://pith.science/pith/OJHYRWXLFEBEWU7C7BAAL727BM","download_json":"https://pith.science/pith/OJHYRWXLFEBEWU7C7BAAL727BM.json","view_paper":"https://pith.science/paper/OJHYRWXL","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.07169&json=true","fetch_graph":"https://pith.science/api/pith-number/OJHYRWXLFEBEWU7C7BAAL727BM/graph.json","fetch_events":"https://pith.science/api/pith-number/OJHYRWXLFEBEWU7C7BAAL727BM/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/OJHYRWXLFEBEWU7C7BAAL727BM/action/timestamp_anchor","attest_storage":"https://pith.science/pith/OJHYRWXLFEBEWU7C7BAAL727BM/action/storage_attestation","attest_author":"https://pith.science/pith/OJHYRWXLFEBEWU7C7BAAL727BM/action/author_attestation","sign_citation":"https://pith.science/pith/OJHYRWXLFEBEWU7C7BAAL727BM/action/citation_signature","submit_replication":"https://pith.science/pith/OJHYRWXLFEBEWU7C7BAAL727BM/action/replication_record"}},"created_at":"2026-07-05T10:12:49.412170+00:00","updated_at":"2026-07-05T10:12:49.412170+00:00"}