{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:FW4BIQPITPOXNYPVVJINUZMZWM","short_pith_number":"pith:FW4BIQPI","schema_version":"1.0","canonical_sha256":"2db81441e89bdd76e1f5aa50da6599b3019e752cd93e24f25317f63b52926136","source":{"kind":"arxiv","id":"2505.00337","version":1},"attestation_state":"computed","paper":{"title":"T2VPhysBench: A First-Principles Benchmark for Physical Consistency in Text-to-Video Generation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.CV"],"primary_cat":"cs.LG","authors_text":"Jiahao Zhang, Jiale Zhao, Jiayan Huo, Xuyang Guo, Zhao Song, Zhenmei Shi","submitted_at":"2025-05-01T06:34:55Z","abstract_excerpt":"Text-to-video generative models have made significant strides in recent years, producing high-quality videos that excel in both aesthetic appeal and accurate instruction following, and have become central to digital art creation and user engagement online. Yet, despite these advancements, their ability to respect fundamental physical laws remains largely untested: many outputs still violate basic constraints such as rigid-body collisions, energy conservation, and gravitational dynamics, resulting in unrealistic or even misleading content. Existing physical-evaluation benchmarks typically rely "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.00337","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-05-01T06:34:55Z","cross_cats_sorted":["cs.AI","cs.CL","cs.CV"],"title_canon_sha256":"e458e2a9038841c91d32ad9f7fbbb94c4acff2d6905050a3f3fbe2e028aec3b1","abstract_canon_sha256":"934371aa6459ef7b068419a3d9627d5d14f86ee17fc282fcb7efa48aa5d95f6b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:57:11.802111Z","signature_b64":"K5I2o9rOBZ7PoTeG10HdgDy2JXdjoFZsclg2uYVIDERsAH+Nkbq1sHhIkSlgGqrDy1cdEEtLx1lgIojMPs4UDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"2db81441e89bdd76e1f5aa50da6599b3019e752cd93e24f25317f63b52926136","last_reissued_at":"2026-07-05T10:57:11.801619Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:57:11.801619Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"T2VPhysBench: A First-Principles Benchmark for Physical Consistency in Text-to-Video Generation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.CV"],"primary_cat":"cs.LG","authors_text":"Jiahao Zhang, Jiale Zhao, Jiayan Huo, Xuyang Guo, Zhao Song, Zhenmei Shi","submitted_at":"2025-05-01T06:34:55Z","abstract_excerpt":"Text-to-video generative models have made significant strides in recent years, producing high-quality videos that excel in both aesthetic appeal and accurate instruction following, and have become central to digital art creation and user engagement online. Yet, despite these advancements, their ability to respect fundamental physical laws remains largely untested: many outputs still violate basic constraints such as rigid-body collisions, energy conservation, and gravitational dynamics, resulting in unrealistic or even misleading content. Existing physical-evaluation benchmarks typically rely "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.00337","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.00337/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.00337","created_at":"2026-07-05T10:57:11.801677+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.00337v1","created_at":"2026-07-05T10:57:11.801677+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.00337","created_at":"2026-07-05T10:57:11.801677+00:00"},{"alias_kind":"pith_short_12","alias_value":"FW4BIQPITPOX","created_at":"2026-07-05T10:57:11.801677+00:00"},{"alias_kind":"pith_short_16","alias_value":"FW4BIQPITPOXNYPV","created_at":"2026-07-05T10:57:11.801677+00:00"},{"alias_kind":"pith_short_8","alias_value":"FW4BIQPI","created_at":"2026-07-05T10:57:11.801677+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":11,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.26969","citing_title":"Einstein World Models","ref_index":64,"is_internal_anchor":false},{"citing_arxiv_id":"2606.20545","citing_title":"Current World Models Lack a Persistent State Core","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07061","citing_title":"Do Joint Audio-Video Generation Models Understand Physics?","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2605.30542","citing_title":"Physically Viable World Models: A Case for Query-Conditioned Embodied AI","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2606.00793","citing_title":"MBench: A Comprehensive Benchmark on Memory Capability for Video World Models","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2605.23699","citing_title":"CRONOS: Benchmarking Counterfactual Physical Consistency in Video Models","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18303","citing_title":"PH-Dreamer: A Physics-Driven World Model via Port-Hamiltonian Generative Dynamics","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2507.00990","citing_title":"Robotic Manipulation by Imitating Generated Videos Without Physical Demonstrations","ref_index":39,"is_internal_anchor":false},{"citing_arxiv_id":"2511.00062","citing_title":"World Simulation with Video Foundation Models for Physical AI","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10806","citing_title":"PhyGround: Benchmarking Physical Reasoning in Generative World Models","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07061","citing_title":"Do Joint Audio-Video Generation Models Understand Physics?","ref_index":14,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/FW4BIQPITPOXNYPVVJINUZMZWM","json":"https://pith.science/pith/FW4BIQPITPOXNYPVVJINUZMZWM.json","graph_json":"https://pith.science/api/pith-number/FW4BIQPITPOXNYPVVJINUZMZWM/graph.json","events_json":"https://pith.science/api/pith-number/FW4BIQPITPOXNYPVVJINUZMZWM/events.json","paper":"https://pith.science/paper/FW4BIQPI"},"agent_actions":{"view_html":"https://pith.science/pith/FW4BIQPITPOXNYPVVJINUZMZWM","download_json":"https://pith.science/pith/FW4BIQPITPOXNYPVVJINUZMZWM.json","view_paper":"https://pith.science/paper/FW4BIQPI","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.00337&json=true","fetch_graph":"https://pith.science/api/pith-number/FW4BIQPITPOXNYPVVJINUZMZWM/graph.json","fetch_events":"https://pith.science/api/pith-number/FW4BIQPITPOXNYPVVJINUZMZWM/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/FW4BIQPITPOXNYPVVJINUZMZWM/action/timestamp_anchor","attest_storage":"https://pith.science/pith/FW4BIQPITPOXNYPVVJINUZMZWM/action/storage_attestation","attest_author":"https://pith.science/pith/FW4BIQPITPOXNYPVVJINUZMZWM/action/author_attestation","sign_citation":"https://pith.science/pith/FW4BIQPITPOXNYPVVJINUZMZWM/action/citation_signature","submit_replication":"https://pith.science/pith/FW4BIQPITPOXNYPVVJINUZMZWM/action/replication_record"}},"created_at":"2026-07-05T10:57:11.801677+00:00","updated_at":"2026-07-05T10:57:11.801677+00:00"}