{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:CXGD6G4VRF7EPNU3W5T6MJB6C6","short_pith_number":"pith:CXGD6G4V","schema_version":"1.0","canonical_sha256":"15cc3f1b95897e47b69bb767e6243e17bc5335ef67060b4218fe26a7e3250e75","source":{"kind":"arxiv","id":"2502.11831","version":1},"attestation_state":"computed","paper":{"title":"Intuitive physics understanding emerges from self-supervised pretraining on natural videos","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Adrien Bardes, Emmanuel Dupoux, Laurent Najman, Mahmoud Assran, Michael Rabbat, Nicolas Ballas, Quentin Garrido, Yann LeCun","submitted_at":"2025-02-17T14:27:14Z","abstract_excerpt":"We investigate the emergence of intuitive physics understanding in general-purpose deep neural network models trained to predict masked regions in natural videos. Leveraging the violation-of-expectation framework, we find that video prediction models trained to predict outcomes in a learned representation space demonstrate an understanding of various intuitive physics properties, such as object permanence and shape consistency. In contrast, video prediction in pixel space and multimodal large language models, which reason through text, achieve performance closer to chance. Our comparisons of t"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.11831","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2025-02-17T14:27:14Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"ae0495e1bffce49e23e0bda91cd0042585569b371c8df3c3226ddd62a4be9cf2","abstract_canon_sha256":"ba3e8aae61863fb0ac859c6063fd81188b36882cd6a292f57f0628e61bf85e1e"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:15:34.851266Z","signature_b64":"9L+xk/J/GtBfqelQdkaqkoG4c/MosK+2wZTb+49yyBUV/wVhOoAIEPqE8b5gYC551W1q3TVLLN701HFdKZpbAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"15cc3f1b95897e47b69bb767e6243e17bc5335ef67060b4218fe26a7e3250e75","last_reissued_at":"2026-07-05T10:15:34.850752Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:15:34.850752Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Intuitive physics understanding emerges from self-supervised pretraining on natural videos","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Adrien Bardes, Emmanuel Dupoux, Laurent Najman, Mahmoud Assran, Michael Rabbat, Nicolas Ballas, Quentin Garrido, Yann LeCun","submitted_at":"2025-02-17T14:27:14Z","abstract_excerpt":"We investigate the emergence of intuitive physics understanding in general-purpose deep neural network models trained to predict masked regions in natural videos. Leveraging the violation-of-expectation framework, we find that video prediction models trained to predict outcomes in a learned representation space demonstrate an understanding of various intuitive physics properties, such as object permanence and shape consistency. In contrast, video prediction in pixel space and multimodal large language models, which reason through text, achieve performance closer to chance. Our comparisons of t"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.11831","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.11831/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.11831","created_at":"2026-07-05T10:15:34.850818+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.11831v1","created_at":"2026-07-05T10:15:34.850818+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.11831","created_at":"2026-07-05T10:15:34.850818+00:00"},{"alias_kind":"pith_short_12","alias_value":"CXGD6G4VRF7E","created_at":"2026-07-05T10:15:34.850818+00:00"},{"alias_kind":"pith_short_16","alias_value":"CXGD6G4VRF7EPNU3","created_at":"2026-07-05T10:15:34.850818+00:00"},{"alias_kind":"pith_short_8","alias_value":"CXGD6G4V","created_at":"2026-07-05T10:15:34.850818+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":16,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.26410","citing_title":"Neural Voxel Dynamics: Learning Implicit 3D Physics via Volumetric Feature Advection","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2606.23153","citing_title":"Asymmetric physics enables efficient learning in quadrupedal robot swarms","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2606.09646","citing_title":"Do Video Foundation Models Understand Intuitive Physics? A Layerwise Probing Analysis","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2606.05497","citing_title":"LEVANTE-bench: Multi-Scale Comparison of VLMs to Children Using Cognitive Tasks (or, \"Is Your VLM Smarter Than a 5th Grader?\")","ref_index":56,"is_internal_anchor":false},{"citing_arxiv_id":"2605.30346","citing_title":"YoCausal: How Far is Video Generation from World Model? A Causality Perspective","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2605.23878","citing_title":"LaMo: Self-Supervised Latent Motion Priors for Physical Realism in Video Generation","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2602.13294","citing_title":"VisPhyWorld: Probing Physical Reasoning via Code-Driven Video Reconstruction","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2604.08503","citing_title":"Phantom: Physics-Infused Video Generation via Joint Modeling of Visual and Latent Physical Dynamics","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2605.15618","citing_title":"Latent Video Prediction Learns Better World Models","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2511.04670","citing_title":"Cambrian-S: Towards Spatial Supersensing in Video","ref_index":44,"is_internal_anchor":false},{"citing_arxiv_id":"2604.03266","citing_title":"Emergent Compositional Communication for Latent World Properties","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2603.19312","citing_title":"LeWorldModel: Stable End-to-End Joint-Embedding Predictive Architecture from Pixels","ref_index":46,"is_internal_anchor":false},{"citing_arxiv_id":"2509.20328","citing_title":"Video models are zero-shot learners and reasoners","ref_index":51,"is_internal_anchor":false},{"citing_arxiv_id":"2604.21192","citing_title":"How VLAs (Really) Work In Open-World Environments","ref_index":36,"is_internal_anchor":false},{"citing_arxiv_id":"2604.08503","citing_title":"Phantom: Physics-Infused Video Generation via Joint Modeling of Visual and Latent Physical Dynamics","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2604.14816","citing_title":"NTIRE 2026 Challenge on Video Saliency Prediction: Methods and Results","ref_index":23,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/CXGD6G4VRF7EPNU3W5T6MJB6C6","json":"https://pith.science/pith/CXGD6G4VRF7EPNU3W5T6MJB6C6.json","graph_json":"https://pith.science/api/pith-number/CXGD6G4VRF7EPNU3W5T6MJB6C6/graph.json","events_json":"https://pith.science/api/pith-number/CXGD6G4VRF7EPNU3W5T6MJB6C6/events.json","paper":"https://pith.science/paper/CXGD6G4V"},"agent_actions":{"view_html":"https://pith.science/pith/CXGD6G4VRF7EPNU3W5T6MJB6C6","download_json":"https://pith.science/pith/CXGD6G4VRF7EPNU3W5T6MJB6C6.json","view_paper":"https://pith.science/paper/CXGD6G4V","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.11831&json=true","fetch_graph":"https://pith.science/api/pith-number/CXGD6G4VRF7EPNU3W5T6MJB6C6/graph.json","fetch_events":"https://pith.science/api/pith-number/CXGD6G4VRF7EPNU3W5T6MJB6C6/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/CXGD6G4VRF7EPNU3W5T6MJB6C6/action/timestamp_anchor","attest_storage":"https://pith.science/pith/CXGD6G4VRF7EPNU3W5T6MJB6C6/action/storage_attestation","attest_author":"https://pith.science/pith/CXGD6G4VRF7EPNU3W5T6MJB6C6/action/author_attestation","sign_citation":"https://pith.science/pith/CXGD6G4VRF7EPNU3W5T6MJB6C6/action/citation_signature","submit_replication":"https://pith.science/pith/CXGD6G4VRF7EPNU3W5T6MJB6C6/action/replication_record"}},"created_at":"2026-07-05T10:15:34.850818+00:00","updated_at":"2026-07-05T10:15:34.850818+00:00"}