{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:MQD7WGKTNXM6FAP4LZ7E26KSGL","short_pith_number":"pith:MQD7WGKT","schema_version":"1.0","canonical_sha256":"6407fb19536dd9e281fc5e7e4d795232f2f61777c8c352a31c2398e92e7c6716","source":{"kind":"arxiv","id":"2507.07202","version":1},"attestation_state":"computed","paper":{"title":"A Survey on Long-Video Storytelling Generation: Architectures, Consistency, and Cinematic Quality","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Abel Salinas, Anh Totti Nguyen, Chien Nguyen, Daksh Dangi, Dinesh Manocha, Eslam Bakr, Franck Dernoncourt, Gang Wu, Hoda Eldardiry, Hongjie Chen, Jaemin Cho, Joe Barrow, Mohamed Elhoseiny, Mohamed Elmoghany, Mohammad Taesiri, Namyong Park, Nedim Lipka, Nesreen Ahmed, Puneet Mathur, Ruiyi Zhang, Ryan Rossi, Seunghyun Yoon, Subhojyoti Mukherjee, Thien Nguyen, Varun Manjunatha, Viet Dac Lai, Xiaolei Huang, Yu Wang, Zhengzhong Tu","submitted_at":"2025-07-09T18:20:33Z","abstract_excerpt":"Despite the significant progress that has been made in video generative models, existing state-of-the-art methods can only produce videos lasting 5-16 seconds, often labeled \"long-form videos\". Furthermore, videos exceeding 16 seconds struggle to maintain consistent character appearances and scene layouts throughout the narrative. In particular, multi-subject long videos still fail to preserve character consistency and motion coherence. While some methods can generate videos up to 150 seconds long, they often suffer from frame redundancy and low temporal diversity. Recent work has attempted to"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2507.07202","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2025-07-09T18:20:33Z","cross_cats_sorted":[],"title_canon_sha256":"8c2d3e0cd6fd7b1ac5c4ccbb298349e7eb8cc29f47dadf95f252b64aa9d9dd96","abstract_canon_sha256":"c9cc64225b4da27466e64edef2dca24c07a4cac5f6d5480c1bcb454e4bd3f8ef"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:34:32.718542Z","signature_b64":"vnXJVHu97vNw56R+eHKqYnYRYkGuY9oeGPbCx5HfJvsaOuLAZgE9LELYrQNJVx3zcd21ghsWPIBM8RnVIgYWCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"6407fb19536dd9e281fc5e7e4d795232f2f61777c8c352a31c2398e92e7c6716","last_reissued_at":"2026-07-05T11:34:32.718052Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:34:32.718052Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"A Survey on Long-Video Storytelling Generation: Architectures, Consistency, and Cinematic Quality","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Abel Salinas, Anh Totti Nguyen, Chien Nguyen, Daksh Dangi, Dinesh Manocha, Eslam Bakr, Franck Dernoncourt, Gang Wu, Hoda Eldardiry, Hongjie Chen, Jaemin Cho, Joe Barrow, Mohamed Elhoseiny, Mohamed Elmoghany, Mohammad Taesiri, Namyong Park, Nedim Lipka, Nesreen Ahmed, Puneet Mathur, Ruiyi Zhang, Ryan Rossi, Seunghyun Yoon, Subhojyoti Mukherjee, Thien Nguyen, Varun Manjunatha, Viet Dac Lai, Xiaolei Huang, Yu Wang, Zhengzhong Tu","submitted_at":"2025-07-09T18:20:33Z","abstract_excerpt":"Despite the significant progress that has been made in video generative models, existing state-of-the-art methods can only produce videos lasting 5-16 seconds, often labeled \"long-form videos\". Furthermore, videos exceeding 16 seconds struggle to maintain consistent character appearances and scene layouts throughout the narrative. In particular, multi-subject long videos still fail to preserve character consistency and motion coherence. While some methods can generate videos up to 150 seconds long, they often suffer from frame redundancy and low temporal diversity. Recent work has attempted to"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2507.07202","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2507.07202/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2507.07202","created_at":"2026-07-05T11:34:32.718116+00:00"},{"alias_kind":"arxiv_version","alias_value":"2507.07202v1","created_at":"2026-07-05T11:34:32.718116+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2507.07202","created_at":"2026-07-05T11:34:32.718116+00:00"},{"alias_kind":"pith_short_12","alias_value":"MQD7WGKTNXM6","created_at":"2026-07-05T11:34:32.718116+00:00"},{"alias_kind":"pith_short_16","alias_value":"MQD7WGKTNXM6FAP4","created_at":"2026-07-05T11:34:32.718116+00:00"},{"alias_kind":"pith_short_8","alias_value":"MQD7WGKT","created_at":"2026-07-05T11:34:32.718116+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/MQD7WGKTNXM6FAP4LZ7E26KSGL","json":"https://pith.science/pith/MQD7WGKTNXM6FAP4LZ7E26KSGL.json","graph_json":"https://pith.science/api/pith-number/MQD7WGKTNXM6FAP4LZ7E26KSGL/graph.json","events_json":"https://pith.science/api/pith-number/MQD7WGKTNXM6FAP4LZ7E26KSGL/events.json","paper":"https://pith.science/paper/MQD7WGKT"},"agent_actions":{"view_html":"https://pith.science/pith/MQD7WGKTNXM6FAP4LZ7E26KSGL","download_json":"https://pith.science/pith/MQD7WGKTNXM6FAP4LZ7E26KSGL.json","view_paper":"https://pith.science/paper/MQD7WGKT","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2507.07202&json=true","fetch_graph":"https://pith.science/api/pith-number/MQD7WGKTNXM6FAP4LZ7E26KSGL/graph.json","fetch_events":"https://pith.science/api/pith-number/MQD7WGKTNXM6FAP4LZ7E26KSGL/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/MQD7WGKTNXM6FAP4LZ7E26KSGL/action/timestamp_anchor","attest_storage":"https://pith.science/pith/MQD7WGKTNXM6FAP4LZ7E26KSGL/action/storage_attestation","attest_author":"https://pith.science/pith/MQD7WGKTNXM6FAP4LZ7E26KSGL/action/author_attestation","sign_citation":"https://pith.science/pith/MQD7WGKTNXM6FAP4LZ7E26KSGL/action/citation_signature","submit_replication":"https://pith.science/pith/MQD7WGKTNXM6FAP4LZ7E26KSGL/action/replication_record"}},"created_at":"2026-07-05T11:34:32.718116+00:00","updated_at":"2026-07-05T11:34:32.718116+00:00"}