{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:4UR2GGKF55OXV5CK43YDWMTYSZ","short_pith_number":"pith:4UR2GGKF","schema_version":"1.0","canonical_sha256":"e523a31945ef5d7af44ae6f03b3278965506d0478d594e73f74331abbd586a4a","source":{"kind":"arxiv","id":"2505.23656","version":1},"attestation_state":"computed","paper":{"title":"VideoREPA: Learning Physics for Video Generation through Relational Alignment with Foundation Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Fanqing Meng, Jiaqi Liao, Junchi Yan, Shaofeng Zhang, Xiangdong Zhang, Xiangpeng Wan, Yu Cheng","submitted_at":"2025-05-29T17:06:44Z","abstract_excerpt":"Recent advancements in text-to-video (T2V) diffusion models have enabled high-fidelity and realistic video synthesis. However, current T2V models often struggle to generate physically plausible content due to their limited inherent ability to accurately understand physics. We found that while the representations within T2V models possess some capacity for physics understanding, they lag significantly behind those from recent video self-supervised learning methods. To this end, we propose a novel framework called VideoREPA, which distills physics understanding capability from video understandin"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.23656","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2025-05-29T17:06:44Z","cross_cats_sorted":[],"title_canon_sha256":"16effa2873d47173c81435a234a974cafa509dbb58e89c9efbb6e6ca0844834b","abstract_canon_sha256":"14c3936356568faec243a62aa81e0f24f3963841f1d614f98076adb77a044d8e"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:12:05.626664Z","signature_b64":"0g5E0x3+eu1M+F4btxxXQHT7uWV50RmauTy5NbupM8A/TrPsTRs3fDUu9LMm3MrnjIae6j11U/fW/beyeHqjAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"e523a31945ef5d7af44ae6f03b3278965506d0478d594e73f74331abbd586a4a","last_reissued_at":"2026-07-05T11:12:05.626060Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:12:05.626060Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"VideoREPA: Learning Physics for Video Generation through Relational Alignment with Foundation Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Fanqing Meng, Jiaqi Liao, Junchi Yan, Shaofeng Zhang, Xiangdong Zhang, Xiangpeng Wan, Yu Cheng","submitted_at":"2025-05-29T17:06:44Z","abstract_excerpt":"Recent advancements in text-to-video (T2V) diffusion models have enabled high-fidelity and realistic video synthesis. However, current T2V models often struggle to generate physically plausible content due to their limited inherent ability to accurately understand physics. We found that while the representations within T2V models possess some capacity for physics understanding, they lag significantly behind those from recent video self-supervised learning methods. To this end, we propose a novel framework called VideoREPA, which distills physics understanding capability from video understandin"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.23656","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.23656/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.23656","created_at":"2026-07-05T11:12:05.626140+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.23656v1","created_at":"2026-07-05T11:12:05.626140+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.23656","created_at":"2026-07-05T11:12:05.626140+00:00"},{"alias_kind":"pith_short_12","alias_value":"4UR2GGKF55OX","created_at":"2026-07-05T11:12:05.626140+00:00"},{"alias_kind":"pith_short_16","alias_value":"4UR2GGKF55OXV5CK","created_at":"2026-07-05T11:12:05.626140+00:00"},{"alias_kind":"pith_short_8","alias_value":"4UR2GGKF","created_at":"2026-07-05T11:12:05.626140+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":21,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.24888","citing_title":"DiffusionBench: On Holistic Evaluation of Diffusion Transformers","ref_index":91,"is_internal_anchor":false},{"citing_arxiv_id":"2606.26916","citing_title":"PhysRAG: Enhancing Physics-Awareness in Video Generation via Retrieval-Augmented Generation","ref_index":90,"is_internal_anchor":false},{"citing_arxiv_id":"2606.17800","citing_title":"MaineCoon: Pursuing A Real-Time Audio-Visual Social World Model","ref_index":65,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01896","citing_title":"Divide and Conquer: Decoupled Representation Alignment for Multimodal World Models","ref_index":61,"is_internal_anchor":false},{"citing_arxiv_id":"2606.04737","citing_title":"Physics-Informed Video Generation via Mixture-of-Experts Latent Alignment","ref_index":40,"is_internal_anchor":false},{"citing_arxiv_id":"2606.02564","citing_title":"VLMs are Good Teachers for Video Reasoning via Adaptive Test-Time Optimization","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07800","citing_title":"SARA: Semantically Adaptive Relational Alignment for Video Diffusion Models","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2605.22882","citing_title":"GEM-4D: Geometry-Enhanced Video World Models for Robot Manipulation","ref_index":68,"is_internal_anchor":false},{"citing_arxiv_id":"2605.24962","citing_title":"Tempered Self-Similarity Alignment for Physically Plausible Video Generation","ref_index":62,"is_internal_anchor":false},{"citing_arxiv_id":"2606.02564","citing_title":"VLMs are Good Teachers for Video Reasoning via Adaptive Test-Time Optimization","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2606.28128","citing_title":"PhysisForcing: Physics Reinforced World Simulator for Robotic Manipulation","ref_index":55,"is_internal_anchor":false},{"citing_arxiv_id":"2605.22882","citing_title":"GEM-4D: Geometry-Enhanced Video World Models for Robot Manipulation","ref_index":69,"is_internal_anchor":false},{"citing_arxiv_id":"2512.01843","citing_title":"PhyDetEx: Detecting and Explaining the Physical Plausibility of T2V Models","ref_index":60,"is_internal_anchor":false},{"citing_arxiv_id":"2605.20808","citing_title":"Spatial Gram Alignment for Ultra-High-Resolution Image Synthesis","ref_index":40,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18324","citing_title":"Improved Baselines with Representation Autoencoders","ref_index":68,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18365","citing_title":"GeoFlow: Enforcing Implicit Geometric Consistency in Video Generation","ref_index":92,"is_internal_anchor":false},{"citing_arxiv_id":"2507.07982","citing_title":"Geometry Forcing: Marrying Video Diffusion and 3D Representation for Consistent World Modeling","ref_index":94,"is_internal_anchor":false},{"citing_arxiv_id":"2510.04978","citing_title":"Aligning Perception, Reasoning, Modeling and Interaction: A Survey on Physical AI","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01896","citing_title":"Divide and Conquer: Decoupled Representation Alignment for Multimodal World Models","ref_index":61,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07800","citing_title":"SARA: Semantically Adaptive Relational Alignment for Video Diffusion Models","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2604.16592","citing_title":"Human Cognition in Machines: A Unified Perspective of World Models","ref_index":221,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/4UR2GGKF55OXV5CK43YDWMTYSZ","json":"https://pith.science/pith/4UR2GGKF55OXV5CK43YDWMTYSZ.json","graph_json":"https://pith.science/api/pith-number/4UR2GGKF55OXV5CK43YDWMTYSZ/graph.json","events_json":"https://pith.science/api/pith-number/4UR2GGKF55OXV5CK43YDWMTYSZ/events.json","paper":"https://pith.science/paper/4UR2GGKF"},"agent_actions":{"view_html":"https://pith.science/pith/4UR2GGKF55OXV5CK43YDWMTYSZ","download_json":"https://pith.science/pith/4UR2GGKF55OXV5CK43YDWMTYSZ.json","view_paper":"https://pith.science/paper/4UR2GGKF","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.23656&json=true","fetch_graph":"https://pith.science/api/pith-number/4UR2GGKF55OXV5CK43YDWMTYSZ/graph.json","fetch_events":"https://pith.science/api/pith-number/4UR2GGKF55OXV5CK43YDWMTYSZ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/4UR2GGKF55OXV5CK43YDWMTYSZ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/4UR2GGKF55OXV5CK43YDWMTYSZ/action/storage_attestation","attest_author":"https://pith.science/pith/4UR2GGKF55OXV5CK43YDWMTYSZ/action/author_attestation","sign_citation":"https://pith.science/pith/4UR2GGKF55OXV5CK43YDWMTYSZ/action/citation_signature","submit_replication":"https://pith.science/pith/4UR2GGKF55OXV5CK43YDWMTYSZ/action/replication_record"}},"created_at":"2026-07-05T11:12:05.626140+00:00","updated_at":"2026-07-05T11:12:05.626140+00:00"}