{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:IOO2J7GWYQPU26JIXSKN2SZRNI","short_pith_number":"pith:IOO2J7GW","schema_version":"1.0","canonical_sha256":"439da4fcd6c41f4d7928bc94dd4b316a041b9f9dc02933026a864aa1e3c12f56","source":{"kind":"arxiv","id":"2503.23368","version":3},"attestation_state":"computed","paper":{"title":"VLIPP: Towards Physically Plausible Video Generation with Vision and Language Informed Physical Prior","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Baolu Li, Huchuan Lu, Jianfei Cai, Lei Bai, Liqian Ma, Tien-Tsin Wong, Xindi Yang, Xu Jia, Yiming Zhang, Zhenfei Yin, Zhiyong Wang","submitted_at":"2025-03-30T09:03:09Z","abstract_excerpt":"Video diffusion models (VDMs) have advanced significantly in recent years, enabling the generation of highly realistic videos and drawing the attention of the community in their potential as world simulators. However, despite their capabilities, VDMs often fail to produce physically plausible videos due to an inherent lack of understanding of physics, resulting in incorrect dynamics and event sequences. To address this limitation, we propose a novel two-stage image-to-video generation framework that explicitly incorporates physics with vision and language informed physical prior. In the first "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2503.23368","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.CV","submitted_at":"2025-03-30T09:03:09Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"c6aecad14a08dec0611180806685fa85a9ebf6c8c508b90f3b0b1a29dc7f739a","abstract_canon_sha256":"bbdf2e938eec70eb11509b048dbd84ea8aa019370d94729ade3e560757ed2d53"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:44:23.455782Z","signature_b64":"y1fySgxBXy0HlxQ2yKGT7/bt6TQGZLRoGbqD3N7gJYOh39Iec4AZQnwJbUWx/3vA8PKwfsHrrlY5mqfMG+CWDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"439da4fcd6c41f4d7928bc94dd4b316a041b9f9dc02933026a864aa1e3c12f56","last_reissued_at":"2026-07-05T10:44:23.455299Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:44:23.455299Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"VLIPP: Towards Physically Plausible Video Generation with Vision and Language Informed Physical Prior","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Baolu Li, Huchuan Lu, Jianfei Cai, Lei Bai, Liqian Ma, Tien-Tsin Wong, Xindi Yang, Xu Jia, Yiming Zhang, Zhenfei Yin, Zhiyong Wang","submitted_at":"2025-03-30T09:03:09Z","abstract_excerpt":"Video diffusion models (VDMs) have advanced significantly in recent years, enabling the generation of highly realistic videos and drawing the attention of the community in their potential as world simulators. However, despite their capabilities, VDMs often fail to produce physically plausible videos due to an inherent lack of understanding of physics, resulting in incorrect dynamics and event sequences. To address this limitation, we propose a novel two-stage image-to-video generation framework that explicitly incorporates physics with vision and language informed physical prior. In the first "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2503.23368","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2503.23368/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2503.23368","created_at":"2026-07-05T10:44:23.455365+00:00"},{"alias_kind":"arxiv_version","alias_value":"2503.23368v3","created_at":"2026-07-05T10:44:23.455365+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2503.23368","created_at":"2026-07-05T10:44:23.455365+00:00"},{"alias_kind":"pith_short_12","alias_value":"IOO2J7GWYQPU","created_at":"2026-07-05T10:44:23.455365+00:00"},{"alias_kind":"pith_short_16","alias_value":"IOO2J7GWYQPU26JI","created_at":"2026-07-05T10:44:23.455365+00:00"},{"alias_kind":"pith_short_8","alias_value":"IOO2J7GW","created_at":"2026-07-05T10:44:23.455365+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.25306","citing_title":"Physics Question Scene Graph: Fine-grained Evaluation of Physical Plausibility in Text-to-Video Generation","ref_index":54,"is_internal_anchor":false},{"citing_arxiv_id":"2606.00499","citing_title":"OptiWorld: Optimal Control for Video World Generation under Physical Constraints","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2601.18577","citing_title":"Self-Refining Video Sampling","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2509.24702","citing_title":"Enhancing Physical Plausibility in Video Generation by Reasoning the Implausibility","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10564","citing_title":"DeepSight: Long-Horizon World Modeling via Latent States Prediction for End-to-End Autonomous Driving","ref_index":137,"is_internal_anchor":false},{"citing_arxiv_id":"2604.06339","citing_title":"Evolution of Video Generative Foundations","ref_index":300,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/IOO2J7GWYQPU26JIXSKN2SZRNI","json":"https://pith.science/pith/IOO2J7GWYQPU26JIXSKN2SZRNI.json","graph_json":"https://pith.science/api/pith-number/IOO2J7GWYQPU26JIXSKN2SZRNI/graph.json","events_json":"https://pith.science/api/pith-number/IOO2J7GWYQPU26JIXSKN2SZRNI/events.json","paper":"https://pith.science/paper/IOO2J7GW"},"agent_actions":{"view_html":"https://pith.science/pith/IOO2J7GWYQPU26JIXSKN2SZRNI","download_json":"https://pith.science/pith/IOO2J7GWYQPU26JIXSKN2SZRNI.json","view_paper":"https://pith.science/paper/IOO2J7GW","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2503.23368&json=true","fetch_graph":"https://pith.science/api/pith-number/IOO2J7GWYQPU26JIXSKN2SZRNI/graph.json","fetch_events":"https://pith.science/api/pith-number/IOO2J7GWYQPU26JIXSKN2SZRNI/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/IOO2J7GWYQPU26JIXSKN2SZRNI/action/timestamp_anchor","attest_storage":"https://pith.science/pith/IOO2J7GWYQPU26JIXSKN2SZRNI/action/storage_attestation","attest_author":"https://pith.science/pith/IOO2J7GWYQPU26JIXSKN2SZRNI/action/author_attestation","sign_citation":"https://pith.science/pith/IOO2J7GWYQPU26JIXSKN2SZRNI/action/citation_signature","submit_replication":"https://pith.science/pith/IOO2J7GWYQPU26JIXSKN2SZRNI/action/replication_record"}},"created_at":"2026-07-05T10:44:23.455365+00:00","updated_at":"2026-07-05T10:44:23.455365+00:00"}