{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:MPJM4L73UFGBYFZSZONP3SZP2Z","short_pith_number":"pith:MPJM4L73","schema_version":"1.0","canonical_sha256":"63d2ce2ffba14c1c1732cb9afdcb2fd6581ec1b9a0cec4a2a8111d78d67309f6","source":{"kind":"arxiv","id":"2506.21552","version":1},"attestation_state":"computed","paper":{"title":"Whole-Body Conditioned Egocentric Video Prediction","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG","cs.MM","cs.RO"],"primary_cat":"cs.CV","authors_text":"Amir Bar, Danny Tran, Jitendra Malik, Trevor Darrell, Yann LeCun, Yutong Bai","submitted_at":"2025-06-26T17:59:59Z","abstract_excerpt":"We train models to Predict Ego-centric Video from human Actions (PEVA), given the past video and an action represented by the relative 3D body pose. By conditioning on kinematic pose trajectories, structured by the joint hierarchy of the body, our model learns to simulate how physical human actions shape the environment from a first-person point of view. We train an auto-regressive conditional diffusion transformer on Nymeria, a large-scale dataset of real-world egocentric video and body pose capture. We further design a hierarchical evaluation protocol with increasingly challenging tasks, ena"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2506.21552","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2025-06-26T17:59:59Z","cross_cats_sorted":["cs.AI","cs.LG","cs.MM","cs.RO"],"title_canon_sha256":"58a524e9827a4e351d38e71adfb03a022816c21bda5adcccf1c5f114ba646c1e","abstract_canon_sha256":"7042fe200e8d5d64ed47df8a640f15e3bb24a3320fbf05e59b5fcec9ec258fae"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:27:47.666021Z","signature_b64":"d9MYrG1SmQggkzk360AOSj8xwWaw1wXStGnJAqbB7RoAZAzHu+rUrLYHcsGG0LZLNIkau1mJMilx8NHHVNwRAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"63d2ce2ffba14c1c1732cb9afdcb2fd6581ec1b9a0cec4a2a8111d78d67309f6","last_reissued_at":"2026-07-05T11:27:47.665475Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:27:47.665475Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Whole-Body Conditioned Egocentric Video Prediction","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG","cs.MM","cs.RO"],"primary_cat":"cs.CV","authors_text":"Amir Bar, Danny Tran, Jitendra Malik, Trevor Darrell, Yann LeCun, Yutong Bai","submitted_at":"2025-06-26T17:59:59Z","abstract_excerpt":"We train models to Predict Ego-centric Video from human Actions (PEVA), given the past video and an action represented by the relative 3D body pose. By conditioning on kinematic pose trajectories, structured by the joint hierarchy of the body, our model learns to simulate how physical human actions shape the environment from a first-person point of view. We train an auto-regressive conditional diffusion transformer on Nymeria, a large-scale dataset of real-world egocentric video and body pose capture. We further design a hierarchical evaluation protocol with increasingly challenging tasks, ena"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.21552","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2506.21552/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2506.21552","created_at":"2026-07-05T11:27:47.665536+00:00"},{"alias_kind":"arxiv_version","alias_value":"2506.21552v1","created_at":"2026-07-05T11:27:47.665536+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.21552","created_at":"2026-07-05T11:27:47.665536+00:00"},{"alias_kind":"pith_short_12","alias_value":"MPJM4L73UFGB","created_at":"2026-07-05T11:27:47.665536+00:00"},{"alias_kind":"pith_short_16","alias_value":"MPJM4L73UFGBYFZS","created_at":"2026-07-05T11:27:47.665536+00:00"},{"alias_kind":"pith_short_8","alias_value":"MPJM4L73","created_at":"2026-07-05T11:27:47.665536+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":14,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.08436","citing_title":"EgoWAM: World Action Models Beyond Pixels with In-the-Wild Egocentric Human Data","ref_index":20,"is_internal_anchor":true},{"citing_arxiv_id":"2606.27325","citing_title":"Not All Actions Are Equal: Rethinking Conditioning for Dexterous World Model","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2607.02075","citing_title":"HandsOnWorld: Unconstrained Egocentric Video Generation with Camera-Disentangled Hand Control","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2606.07326","citing_title":"AnchorWorld: Embodied Egocentric World Simulation with View-based Evolution Customization","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2606.04130","citing_title":"CLAW: Learning Continuous Latent Action World Models via Adversarial Latent Regularization","ref_index":42,"is_internal_anchor":false},{"citing_arxiv_id":"2605.15477","citing_title":"EgoExo-WM: Unlocking Exo Video for Ego World Models","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2605.26316","citing_title":"E$^3$C: Video Generation with 3D Environmental Memory and Ego-Exo Human Pose Control","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2605.20388","citing_title":"How You Move Tells What You'll Do: Trajectory-Conditioned Egocentric Prediction","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2605.15477","citing_title":"EgoExo-WM: Unlocking Exo Video for Ego World Models","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2511.04670","citing_title":"Cambrian-S: Towards Spatial Supersensing in Video","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2602.06949","citing_title":"DreamDojo: A Generalist Robot World Model from Large-Scale Human Videos","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2509.24527","citing_title":"Training Agents Inside of Scalable World Models","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2604.03208","citing_title":"Hierarchical Planning with Latent World Models","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12090","citing_title":"World Action Models: The Next Frontier in Embodied AI","ref_index":300,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/MPJM4L73UFGBYFZSZONP3SZP2Z","json":"https://pith.science/pith/MPJM4L73UFGBYFZSZONP3SZP2Z.json","graph_json":"https://pith.science/api/pith-number/MPJM4L73UFGBYFZSZONP3SZP2Z/graph.json","events_json":"https://pith.science/api/pith-number/MPJM4L73UFGBYFZSZONP3SZP2Z/events.json","paper":"https://pith.science/paper/MPJM4L73"},"agent_actions":{"view_html":"https://pith.science/pith/MPJM4L73UFGBYFZSZONP3SZP2Z","download_json":"https://pith.science/pith/MPJM4L73UFGBYFZSZONP3SZP2Z.json","view_paper":"https://pith.science/paper/MPJM4L73","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2506.21552&json=true","fetch_graph":"https://pith.science/api/pith-number/MPJM4L73UFGBYFZSZONP3SZP2Z/graph.json","fetch_events":"https://pith.science/api/pith-number/MPJM4L73UFGBYFZSZONP3SZP2Z/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/MPJM4L73UFGBYFZSZONP3SZP2Z/action/timestamp_anchor","attest_storage":"https://pith.science/pith/MPJM4L73UFGBYFZSZONP3SZP2Z/action/storage_attestation","attest_author":"https://pith.science/pith/MPJM4L73UFGBYFZSZONP3SZP2Z/action/author_attestation","sign_citation":"https://pith.science/pith/MPJM4L73UFGBYFZSZONP3SZP2Z/action/citation_signature","submit_replication":"https://pith.science/pith/MPJM4L73UFGBYFZSZONP3SZP2Z/action/replication_record"}},"created_at":"2026-07-05T11:27:47.665536+00:00","updated_at":"2026-07-05T11:27:47.665536+00:00"}