{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:D7TCONMJPSBIWFADWF4KNFRY44","short_pith_number":"pith:D7TCONMJ","schema_version":"1.0","canonical_sha256":"1fe62735897c828b1403b178a69638e73c5af956b66608e7dbe40ad8df26161e","source":{"kind":"arxiv","id":"2412.16211","version":1},"attestation_state":"computed","paper":{"title":"Is Your World Simulator a Good Story Presenter? A Consecutive Events-Based Benchmark for Future Long Video Generation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL","cs.GR"],"primary_cat":"cs.CV","authors_text":"Jianwei Yang, Kuan Wang, Luyao Ma, Shuohang Wang, Simon Shaolei Du, Xuehai He, Yelong Shen, Yiping Wang","submitted_at":"2024-12-17T23:00:42Z","abstract_excerpt":"The current state-of-the-art video generative models can produce commercial-grade videos with highly realistic details. However, they still struggle to coherently present multiple sequential events in the stories specified by the prompts, which is foreseeable an essential capability for future long video generation scenarios. For example, top T2V generative models still fail to generate a video of the short simple story 'how to put an elephant into a refrigerator.' While existing detail-oriented benchmarks primarily focus on fine-grained metrics like aesthetic quality and spatial-temporal cons"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2412.16211","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-12-17T23:00:42Z","cross_cats_sorted":["cs.CL","cs.GR"],"title_canon_sha256":"32e6edc5b686f30db329f6c9d7d7f7e6e0de9baad99db8d47186f1c83165d276","abstract_canon_sha256":"1ca358f05b4e24edbffbd9e92f6d54cc5612dea9e93d77a411353f6d4c9cc2be"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:52:43.086879Z","signature_b64":"E0mdn8dB6HlCo4rphCyUouX1k2v5GxFoIJqX+mXACO9COee4TRYj4uMIUhhCnYVu76yuBd+awDSzET26ThPHAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"1fe62735897c828b1403b178a69638e73c5af956b66608e7dbe40ad8df26161e","last_reissued_at":"2026-07-05T09:52:43.086417Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:52:43.086417Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Is Your World Simulator a Good Story Presenter? A Consecutive Events-Based Benchmark for Future Long Video Generation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL","cs.GR"],"primary_cat":"cs.CV","authors_text":"Jianwei Yang, Kuan Wang, Luyao Ma, Shuohang Wang, Simon Shaolei Du, Xuehai He, Yelong Shen, Yiping Wang","submitted_at":"2024-12-17T23:00:42Z","abstract_excerpt":"The current state-of-the-art video generative models can produce commercial-grade videos with highly realistic details. However, they still struggle to coherently present multiple sequential events in the stories specified by the prompts, which is foreseeable an essential capability for future long video generation scenarios. For example, top T2V generative models still fail to generate a video of the short simple story 'how to put an elephant into a refrigerator.' While existing detail-oriented benchmarks primarily focus on fine-grained metrics like aesthetic quality and spatial-temporal cons"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2412.16211","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2412.16211/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2412.16211","created_at":"2026-07-05T09:52:43.086475+00:00"},{"alias_kind":"arxiv_version","alias_value":"2412.16211v1","created_at":"2026-07-05T09:52:43.086475+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2412.16211","created_at":"2026-07-05T09:52:43.086475+00:00"},{"alias_kind":"pith_short_12","alias_value":"D7TCONMJPSBI","created_at":"2026-07-05T09:52:43.086475+00:00"},{"alias_kind":"pith_short_16","alias_value":"D7TCONMJPSBIWFAD","created_at":"2026-07-05T09:52:43.086475+00:00"},{"alias_kind":"pith_short_8","alias_value":"D7TCONMJ","created_at":"2026-07-05T09:52:43.086475+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.10543","citing_title":"TIE: Time Interval Encoding for Video Generation over Events","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2503.21755","citing_title":"VBench-2.0: Advancing Video Generation Benchmark Suite for Intrinsic Faithfulness","ref_index":74,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10543","citing_title":"TIE: Time Interval Encoding for Video Generation over Events","ref_index":35,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/D7TCONMJPSBIWFADWF4KNFRY44","json":"https://pith.science/pith/D7TCONMJPSBIWFADWF4KNFRY44.json","graph_json":"https://pith.science/api/pith-number/D7TCONMJPSBIWFADWF4KNFRY44/graph.json","events_json":"https://pith.science/api/pith-number/D7TCONMJPSBIWFADWF4KNFRY44/events.json","paper":"https://pith.science/paper/D7TCONMJ"},"agent_actions":{"view_html":"https://pith.science/pith/D7TCONMJPSBIWFADWF4KNFRY44","download_json":"https://pith.science/pith/D7TCONMJPSBIWFADWF4KNFRY44.json","view_paper":"https://pith.science/paper/D7TCONMJ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2412.16211&json=true","fetch_graph":"https://pith.science/api/pith-number/D7TCONMJPSBIWFADWF4KNFRY44/graph.json","fetch_events":"https://pith.science/api/pith-number/D7TCONMJPSBIWFADWF4KNFRY44/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/D7TCONMJPSBIWFADWF4KNFRY44/action/timestamp_anchor","attest_storage":"https://pith.science/pith/D7TCONMJPSBIWFADWF4KNFRY44/action/storage_attestation","attest_author":"https://pith.science/pith/D7TCONMJPSBIWFADWF4KNFRY44/action/author_attestation","sign_citation":"https://pith.science/pith/D7TCONMJPSBIWFADWF4KNFRY44/action/citation_signature","submit_replication":"https://pith.science/pith/D7TCONMJPSBIWFADWF4KNFRY44/action/replication_record"}},"created_at":"2026-07-05T09:52:43.086475+00:00","updated_at":"2026-07-05T09:52:43.086475+00:00"}