{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:FFGQ5LYHF4Z5RK2EBA3AVMFPTG","short_pith_number":"pith:FFGQ5LYH","schema_version":"1.0","canonical_sha256":"294d0eaf072f33d8ab4408360ab0af998f74f9c375a13f45c59ac1f5bdc0e165","source":{"kind":"arxiv","id":"2507.18107","version":1},"attestation_state":"computed","paper":{"title":"T2VWorldBench: A Benchmark for Evaluating World Knowledge in Text-to-Video Generation","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Jiahao Zhang, Xuyang Guo, Yubin Chen, Zhao Song, Zhenmei Shi","submitted_at":"2025-07-24T05:37:08Z","abstract_excerpt":"Text-to-video (T2V) models have shown remarkable performance in generating visually reasonable scenes, while their capability to leverage world knowledge for ensuring semantic consistency and factual accuracy remains largely understudied. In response to this challenge, we propose T2VWorldBench, the first systematic evaluation framework for evaluating the world knowledge generation abilities of text-to-video models, covering 6 major categories, 60 subcategories, and 1,200 prompts across a wide range of domains, including physics, nature, activity, culture, causality, and object. To address both"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2507.18107","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.CV","submitted_at":"2025-07-24T05:37:08Z","cross_cats_sorted":[],"title_canon_sha256":"62c456091a5bc0656ff0c2fc9015bbde74296245810479418312867450225881","abstract_canon_sha256":"15a7f6c53dd793963ebcc15df593fe16375ea7ca6291fe44fd4e86d094acb1e4"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:42:41.711014Z","signature_b64":"i7ofK+2X7t+wnJKHMXP2yshIBlkHOm6gx/8ZEjHJ2t36xGUwSlfNQ9FHQKt27jbYZB1E3U9hPMHldDfU3XDmBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"294d0eaf072f33d8ab4408360ab0af998f74f9c375a13f45c59ac1f5bdc0e165","last_reissued_at":"2026-07-05T11:42:41.710342Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:42:41.710342Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"T2VWorldBench: A Benchmark for Evaluating World Knowledge in Text-to-Video Generation","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Jiahao Zhang, Xuyang Guo, Yubin Chen, Zhao Song, Zhenmei Shi","submitted_at":"2025-07-24T05:37:08Z","abstract_excerpt":"Text-to-video (T2V) models have shown remarkable performance in generating visually reasonable scenes, while their capability to leverage world knowledge for ensuring semantic consistency and factual accuracy remains largely understudied. In response to this challenge, we propose T2VWorldBench, the first systematic evaluation framework for evaluating the world knowledge generation abilities of text-to-video models, covering 6 major categories, 60 subcategories, and 1,200 prompts across a wide range of domains, including physics, nature, activity, culture, causality, and object. To address both"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2507.18107","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2507.18107/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2507.18107","created_at":"2026-07-05T11:42:41.710426+00:00"},{"alias_kind":"arxiv_version","alias_value":"2507.18107v1","created_at":"2026-07-05T11:42:41.710426+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2507.18107","created_at":"2026-07-05T11:42:41.710426+00:00"},{"alias_kind":"pith_short_12","alias_value":"FFGQ5LYHF4Z5","created_at":"2026-07-05T11:42:41.710426+00:00"},{"alias_kind":"pith_short_16","alias_value":"FFGQ5LYHF4Z5RK2E","created_at":"2026-07-05T11:42:41.710426+00:00"},{"alias_kind":"pith_short_8","alias_value":"FFGQ5LYH","created_at":"2026-07-05T11:42:41.710426+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.16716","citing_title":"When Cultures Move: Measuring and Improving Multicultural Text-to-Video Generation","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2605.16716","citing_title":"When Cultures Move: Measuring and Improving Multicultural Text-to-Video Generation","ref_index":2,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/FFGQ5LYHF4Z5RK2EBA3AVMFPTG","json":"https://pith.science/pith/FFGQ5LYHF4Z5RK2EBA3AVMFPTG.json","graph_json":"https://pith.science/api/pith-number/FFGQ5LYHF4Z5RK2EBA3AVMFPTG/graph.json","events_json":"https://pith.science/api/pith-number/FFGQ5LYHF4Z5RK2EBA3AVMFPTG/events.json","paper":"https://pith.science/paper/FFGQ5LYH"},"agent_actions":{"view_html":"https://pith.science/pith/FFGQ5LYHF4Z5RK2EBA3AVMFPTG","download_json":"https://pith.science/pith/FFGQ5LYHF4Z5RK2EBA3AVMFPTG.json","view_paper":"https://pith.science/paper/FFGQ5LYH","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2507.18107&json=true","fetch_graph":"https://pith.science/api/pith-number/FFGQ5LYHF4Z5RK2EBA3AVMFPTG/graph.json","fetch_events":"https://pith.science/api/pith-number/FFGQ5LYHF4Z5RK2EBA3AVMFPTG/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/FFGQ5LYHF4Z5RK2EBA3AVMFPTG/action/timestamp_anchor","attest_storage":"https://pith.science/pith/FFGQ5LYHF4Z5RK2EBA3AVMFPTG/action/storage_attestation","attest_author":"https://pith.science/pith/FFGQ5LYHF4Z5RK2EBA3AVMFPTG/action/author_attestation","sign_citation":"https://pith.science/pith/FFGQ5LYHF4Z5RK2EBA3AVMFPTG/action/citation_signature","submit_replication":"https://pith.science/pith/FFGQ5LYHF4Z5RK2EBA3AVMFPTG/action/replication_record"}},"created_at":"2026-07-05T11:42:41.710426+00:00","updated_at":"2026-07-05T11:42:41.710426+00:00"}