{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:5QWBL5R35R7XQLYXKNLCDEU4XN","short_pith_number":"pith:5QWBL5R3","schema_version":"1.0","canonical_sha256":"ec2c15f63bec7f782f17535621929cbb79d4b1f1e823fe711d15650e88470337","source":{"kind":"arxiv","id":"2406.00439","version":1},"attestation_state":"computed","paper":{"title":"Learning Manipulation by Predicting Interaction","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CV"],"primary_cat":"cs.RO","authors_text":"Bangjun Wang, Bin Zhao, Di Hu, Dong Wang, Hao Dong, Haoming Song, Heming Cui, Hongyang Li, Jia Zeng, Li Chen, Ping Luo, Qingwen Bu, Wenke Xia, Xuelong Li, Yu Qiao","submitted_at":"2024-06-01T13:28:31Z","abstract_excerpt":"Representation learning approaches for robotic manipulation have boomed in recent years. Due to the scarcity of in-domain robot data, prevailing methodologies tend to leverage large-scale human video datasets to extract generalizable features for visuomotor policy learning. Despite the progress achieved, prior endeavors disregard the interactive dynamics that capture behavior patterns and physical interaction during the manipulation process, resulting in an inadequate understanding of the relationship between objects and the environment. To this end, we propose a general pre-training pipeline "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2406.00439","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.RO","submitted_at":"2024-06-01T13:28:31Z","cross_cats_sorted":["cs.CV"],"title_canon_sha256":"605956b048bd0aa5c43df718ab668d2c3e6611b0ddadcea84d8c38dac2eac000","abstract_canon_sha256":"deb0212ddadd0375d7ba4b208a1b21199c0540efbafec0804b7698771a282f79"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:26:01.971000Z","signature_b64":"KBowENQrUquA2lIMmCSCmla1YTMgwkVZBs2S8/BpcZ4ZhPo2d099pqMPuEBrBiUcOiQquMORrSyfbGXpDZTLDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ec2c15f63bec7f782f17535621929cbb79d4b1f1e823fe711d15650e88470337","last_reissued_at":"2026-07-05T08:26:01.970403Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:26:01.970403Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Learning Manipulation by Predicting Interaction","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CV"],"primary_cat":"cs.RO","authors_text":"Bangjun Wang, Bin Zhao, Di Hu, Dong Wang, Hao Dong, Haoming Song, Heming Cui, Hongyang Li, Jia Zeng, Li Chen, Ping Luo, Qingwen Bu, Wenke Xia, Xuelong Li, Yu Qiao","submitted_at":"2024-06-01T13:28:31Z","abstract_excerpt":"Representation learning approaches for robotic manipulation have boomed in recent years. Due to the scarcity of in-domain robot data, prevailing methodologies tend to leverage large-scale human video datasets to extract generalizable features for visuomotor policy learning. Despite the progress achieved, prior endeavors disregard the interactive dynamics that capture behavior patterns and physical interaction during the manipulation process, resulting in an inadequate understanding of the relationship between objects and the environment. To this end, we propose a general pre-training pipeline "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.00439","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.00439/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2406.00439","created_at":"2026-07-05T08:26:01.970468+00:00"},{"alias_kind":"arxiv_version","alias_value":"2406.00439v1","created_at":"2026-07-05T08:26:01.970468+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.00439","created_at":"2026-07-05T08:26:01.970468+00:00"},{"alias_kind":"pith_short_12","alias_value":"5QWBL5R35R7X","created_at":"2026-07-05T08:26:01.970468+00:00"},{"alias_kind":"pith_short_16","alias_value":"5QWBL5R35R7XQLYX","created_at":"2026-07-05T08:26:01.970468+00:00"},{"alias_kind":"pith_short_8","alias_value":"5QWBL5R3","created_at":"2026-07-05T08:26:01.970468+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.12499","citing_title":"Action-Effect Memory Pretraining for Robot Manipulation","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2606.28215","citing_title":"HAT-4D: Lifting Monocular Video for 4D Multi-Object Interactions via Human-Agent Collaboration","ref_index":50,"is_internal_anchor":false},{"citing_arxiv_id":"2507.12440","citing_title":"EgoVLA: Learning Vision-Language-Action Models from Egocentric Human Videos","ref_index":65,"is_internal_anchor":false},{"citing_arxiv_id":"2505.15659","citing_title":"FLARE: Robot Learning with Implicit World Modeling","ref_index":46,"is_internal_anchor":false},{"citing_arxiv_id":"2601.07060","citing_title":"PALM: Progress-Aware Policy Learning via Affordance Reasoning for Long-Horizon Robotic Manipulation","ref_index":138,"is_internal_anchor":false},{"citing_arxiv_id":"2505.12705","citing_title":"DreamGen: Unlocking Generalization in Robot Learning through Video World Models","ref_index":58,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/5QWBL5R35R7XQLYXKNLCDEU4XN","json":"https://pith.science/pith/5QWBL5R35R7XQLYXKNLCDEU4XN.json","graph_json":"https://pith.science/api/pith-number/5QWBL5R35R7XQLYXKNLCDEU4XN/graph.json","events_json":"https://pith.science/api/pith-number/5QWBL5R35R7XQLYXKNLCDEU4XN/events.json","paper":"https://pith.science/paper/5QWBL5R3"},"agent_actions":{"view_html":"https://pith.science/pith/5QWBL5R35R7XQLYXKNLCDEU4XN","download_json":"https://pith.science/pith/5QWBL5R35R7XQLYXKNLCDEU4XN.json","view_paper":"https://pith.science/paper/5QWBL5R3","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2406.00439&json=true","fetch_graph":"https://pith.science/api/pith-number/5QWBL5R35R7XQLYXKNLCDEU4XN/graph.json","fetch_events":"https://pith.science/api/pith-number/5QWBL5R35R7XQLYXKNLCDEU4XN/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/5QWBL5R35R7XQLYXKNLCDEU4XN/action/timestamp_anchor","attest_storage":"https://pith.science/pith/5QWBL5R35R7XQLYXKNLCDEU4XN/action/storage_attestation","attest_author":"https://pith.science/pith/5QWBL5R35R7XQLYXKNLCDEU4XN/action/author_attestation","sign_citation":"https://pith.science/pith/5QWBL5R35R7XQLYXKNLCDEU4XN/action/citation_signature","submit_replication":"https://pith.science/pith/5QWBL5R35R7XQLYXKNLCDEU4XN/action/replication_record"}},"created_at":"2026-07-05T08:26:01.970468+00:00","updated_at":"2026-07-05T08:26:01.970468+00:00"}