{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:5MVEOHAE5DIPLU5YUSCVH4IDZT","short_pith_number":"pith:5MVEOHAE","schema_version":"1.0","canonical_sha256":"eb2a471c04e8d0f5d3b8a48553f103ccf5e7390a129446df460158e003673139","source":{"kind":"arxiv","id":"2508.19852","version":2},"attestation_state":"computed","paper":{"title":"Ego-centric Predictive Model Conditioned on Hand Trajectories","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Binjie Zhang, Mike Zheng Shou","submitted_at":"2025-08-27T13:09:55Z","abstract_excerpt":"In egocentric scenarios, anticipating both the next action and its visual outcome is essential for understanding human-object interactions and for enabling robotic planning. However, existing paradigms fall short of jointly modeling these aspects. Vision-Language-Action (VLA) models focus on action prediction but lack explicit modeling of how actions influence the visual scene, while video prediction models generate future frames without conditioning on specific actions, often resulting in implausible or contextually inconsistent outcomes. To bridge this gap, we propose a unified two-stage pre"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2508.19852","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2025-08-27T13:09:55Z","cross_cats_sorted":[],"title_canon_sha256":"65f2a1c0de9dcbb6133082a9f7ede193bbfcff676cf3d2531e875ce7039e4df0","abstract_canon_sha256":"a0043a620aea5145a3b032841440bfd13de38740819c6993c541c90a0e128100"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T12:00:52.477960Z","signature_b64":"hWKD0inprYgoH+TMa1zs3DXbEL+pMAN1y6jWU1ZO48CWGOJTxxezVgGkBjes0d4qGCnDVqzQLFEvx2RyGXV7Ag==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"eb2a471c04e8d0f5d3b8a48553f103ccf5e7390a129446df460158e003673139","last_reissued_at":"2026-07-05T12:00:52.477469Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T12:00:52.477469Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Ego-centric Predictive Model Conditioned on Hand Trajectories","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Binjie Zhang, Mike Zheng Shou","submitted_at":"2025-08-27T13:09:55Z","abstract_excerpt":"In egocentric scenarios, anticipating both the next action and its visual outcome is essential for understanding human-object interactions and for enabling robotic planning. However, existing paradigms fall short of jointly modeling these aspects. Vision-Language-Action (VLA) models focus on action prediction but lack explicit modeling of how actions influence the visual scene, while video prediction models generate future frames without conditioning on specific actions, often resulting in implausible or contextually inconsistent outcomes. To bridge this gap, we propose a unified two-stage pre"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2508.19852","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2508.19852/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2508.19852","created_at":"2026-07-05T12:00:52.477523+00:00"},{"alias_kind":"arxiv_version","alias_value":"2508.19852v2","created_at":"2026-07-05T12:00:52.477523+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2508.19852","created_at":"2026-07-05T12:00:52.477523+00:00"},{"alias_kind":"pith_short_12","alias_value":"5MVEOHAE5DIP","created_at":"2026-07-05T12:00:52.477523+00:00"},{"alias_kind":"pith_short_16","alias_value":"5MVEOHAE5DIPLU5Y","created_at":"2026-07-05T12:00:52.477523+00:00"},{"alias_kind":"pith_short_8","alias_value":"5MVEOHAE","created_at":"2026-07-05T12:00:52.477523+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2603.11755","citing_title":"Controllable Egocentric Video Generation via Occlusion-Aware Sparse 3D Hand Joints","ref_index":62,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/5MVEOHAE5DIPLU5YUSCVH4IDZT","json":"https://pith.science/pith/5MVEOHAE5DIPLU5YUSCVH4IDZT.json","graph_json":"https://pith.science/api/pith-number/5MVEOHAE5DIPLU5YUSCVH4IDZT/graph.json","events_json":"https://pith.science/api/pith-number/5MVEOHAE5DIPLU5YUSCVH4IDZT/events.json","paper":"https://pith.science/paper/5MVEOHAE"},"agent_actions":{"view_html":"https://pith.science/pith/5MVEOHAE5DIPLU5YUSCVH4IDZT","download_json":"https://pith.science/pith/5MVEOHAE5DIPLU5YUSCVH4IDZT.json","view_paper":"https://pith.science/paper/5MVEOHAE","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2508.19852&json=true","fetch_graph":"https://pith.science/api/pith-number/5MVEOHAE5DIPLU5YUSCVH4IDZT/graph.json","fetch_events":"https://pith.science/api/pith-number/5MVEOHAE5DIPLU5YUSCVH4IDZT/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/5MVEOHAE5DIPLU5YUSCVH4IDZT/action/timestamp_anchor","attest_storage":"https://pith.science/pith/5MVEOHAE5DIPLU5YUSCVH4IDZT/action/storage_attestation","attest_author":"https://pith.science/pith/5MVEOHAE5DIPLU5YUSCVH4IDZT/action/author_attestation","sign_citation":"https://pith.science/pith/5MVEOHAE5DIPLU5YUSCVH4IDZT/action/citation_signature","submit_replication":"https://pith.science/pith/5MVEOHAE5DIPLU5YUSCVH4IDZT/action/replication_record"}},"created_at":"2026-07-05T12:00:52.477523+00:00","updated_at":"2026-07-05T12:00:52.477523+00:00"}