{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:KO2EJIUPYZZE4NWSG677O5JCP5","short_pith_number":"pith:KO2EJIUP","schema_version":"1.0","canonical_sha256":"53b444a28fc6724e36d237bff775227f6e635093734157a805abd9886e94e3f0","source":{"kind":"arxiv","id":"2309.15091","version":2},"attestation_state":"computed","paper":{"title":"VideoDirectorGPT: Consistent Multi-scene Video Generation via LLM-Guided Planning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.LG"],"primary_cat":"cs.CV","authors_text":"Abhay Zala, Han Lin, Jaemin Cho, Mohit Bansal","submitted_at":"2023-09-26T17:36:26Z","abstract_excerpt":"Recent text-to-video (T2V) generation methods have seen significant advancements. However, the majority of these works focus on producing short video clips of a single event (i.e., single-scene videos). Meanwhile, recent large language models (LLMs) have demonstrated their capability in generating layouts and programs to control downstream visual modules. This prompts an important question: can we leverage the knowledge embedded in these LLMs for temporally consistent long video generation? In this paper, we propose VideoDirectorGPT, a novel framework for consistent multi-scene video generatio"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2309.15091","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2023-09-26T17:36:26Z","cross_cats_sorted":["cs.AI","cs.CL","cs.LG"],"title_canon_sha256":"ed1f9c92447326189a7f2b05ea23dd5388504a2883864b1d71d9f8cbb5db9a14","abstract_canon_sha256":"91e06fdab4f579fec178bdf8cad6f6ddf00535baf459d5a7ab72777a771ce65a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:43:23.423326Z","signature_b64":"JP3WNrpAOa33eNzZzr/AoAbjbHwv9J4SU2ZcWkP9HVE74XXKBw2P23DDaR+nq/jItjBkoMpUD2H50IP3ugR/CQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"53b444a28fc6724e36d237bff775227f6e635093734157a805abd9886e94e3f0","last_reissued_at":"2026-07-05T08:43:23.422840Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:43:23.422840Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"VideoDirectorGPT: Consistent Multi-scene Video Generation via LLM-Guided Planning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.LG"],"primary_cat":"cs.CV","authors_text":"Abhay Zala, Han Lin, Jaemin Cho, Mohit Bansal","submitted_at":"2023-09-26T17:36:26Z","abstract_excerpt":"Recent text-to-video (T2V) generation methods have seen significant advancements. However, the majority of these works focus on producing short video clips of a single event (i.e., single-scene videos). Meanwhile, recent large language models (LLMs) have demonstrated their capability in generating layouts and programs to control downstream visual modules. This prompts an important question: can we leverage the knowledge embedded in these LLMs for temporally consistent long video generation? In this paper, we propose VideoDirectorGPT, a novel framework for consistent multi-scene video generatio"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2309.15091","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2309.15091/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2309.15091","created_at":"2026-07-05T08:43:23.422911+00:00"},{"alias_kind":"arxiv_version","alias_value":"2309.15091v2","created_at":"2026-07-05T08:43:23.422911+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2309.15091","created_at":"2026-07-05T08:43:23.422911+00:00"},{"alias_kind":"pith_short_12","alias_value":"KO2EJIUPYZZE","created_at":"2026-07-05T08:43:23.422911+00:00"},{"alias_kind":"pith_short_16","alias_value":"KO2EJIUPYZZE4NWS","created_at":"2026-07-05T08:43:23.422911+00:00"},{"alias_kind":"pith_short_8","alias_value":"KO2EJIUP","created_at":"2026-07-05T08:43:23.422911+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":20,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.24107","citing_title":"DramaDirector: Geometry-Guided Short Drama Generation","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2606.10620","citing_title":"Can Image Models Imagine Time? ImageTime: A Novel Benchmark for Probing Visual World Modeling Through Spatiotemporal Consistency","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2606.08091","citing_title":"VideoWeaver: Evaluating and Evolving Skills for Agentic Long Video Generation","ref_index":38,"is_internal_anchor":false},{"citing_arxiv_id":"2606.07636","citing_title":"Crayotter: Traceable Multi-Agent Workflows for Long-Form Video Editing","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2605.26244","citing_title":"LongAV-Compass: Towards Unified Evaluation of Minute-Scale Audio-Visual Generation Across T2AV, I2AV, and V2AV","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2605.28035","citing_title":"MTAVG-Bench 2.0: Diagnosing Failure Modes of Cinematic Expressiveness in Multi-Talker Audio-Video Generation","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2411.15115","citing_title":"Self-Correcting Text-to-Video Generation with Misalignment Detection and Localized Refinement","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2503.06310","citing_title":"Scene-Action Prompt Fusion for Coherent Text-to-Video Storytelling","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2505.16819","citing_title":"Character-Centered Dialogue Generation from Scene-Level Prompts","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2605.22144","citing_title":"One Sentence, One Drama: Personalized Short-Form Drama Generation via Multi-Agent Systems","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17423","citing_title":"Soap2Soap: Long Cinematic Video Remaking via Multi-Agent Collaboration","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2605.15199","citing_title":"EntityBench: Towards Entity-Consistent Long-Range Multi-Shot Video Generation","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2604.03315","citing_title":"StoryBlender: Inter-Shot Consistent and Editable 3D Storyboard with Spatial-temporal Dynamics","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11363","citing_title":"PresentAgent-2: Towards Generalist Multimodal Presentation Agents","ref_index":38,"is_internal_anchor":false},{"citing_arxiv_id":"2604.25318","citing_title":"Cutscene Agent: An LLM Agent Framework for Automated 3D Cutscene Generation","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2604.23579","citing_title":"CineAGI: Character-Consistent Movie Creation through LLM-Orchestrated Multi-Modal Generation and Cross-Scene Integration","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2604.10383","citing_title":"Authoring for Living Worlds: Tool-Constrained LLM Agents for Executable Multi-Actor Scenarios","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2604.06339","citing_title":"Evolution of Video Generative Foundations","ref_index":68,"is_internal_anchor":false},{"citing_arxiv_id":"2604.17749","citing_title":"Ego-InBetween: Generating Object State Transitions in Ego-Centric Videos","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2604.19473","citing_title":"TS-Attn: Temporal-wise Separable Attention for Multi-Event Video Generation","ref_index":15,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/KO2EJIUPYZZE4NWSG677O5JCP5","json":"https://pith.science/pith/KO2EJIUPYZZE4NWSG677O5JCP5.json","graph_json":"https://pith.science/api/pith-number/KO2EJIUPYZZE4NWSG677O5JCP5/graph.json","events_json":"https://pith.science/api/pith-number/KO2EJIUPYZZE4NWSG677O5JCP5/events.json","paper":"https://pith.science/paper/KO2EJIUP"},"agent_actions":{"view_html":"https://pith.science/pith/KO2EJIUPYZZE4NWSG677O5JCP5","download_json":"https://pith.science/pith/KO2EJIUPYZZE4NWSG677O5JCP5.json","view_paper":"https://pith.science/paper/KO2EJIUP","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2309.15091&json=true","fetch_graph":"https://pith.science/api/pith-number/KO2EJIUPYZZE4NWSG677O5JCP5/graph.json","fetch_events":"https://pith.science/api/pith-number/KO2EJIUPYZZE4NWSG677O5JCP5/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/KO2EJIUPYZZE4NWSG677O5JCP5/action/timestamp_anchor","attest_storage":"https://pith.science/pith/KO2EJIUPYZZE4NWSG677O5JCP5/action/storage_attestation","attest_author":"https://pith.science/pith/KO2EJIUPYZZE4NWSG677O5JCP5/action/author_attestation","sign_citation":"https://pith.science/pith/KO2EJIUPYZZE4NWSG677O5JCP5/action/citation_signature","submit_replication":"https://pith.science/pith/KO2EJIUPYZZE4NWSG677O5JCP5/action/replication_record"}},"created_at":"2026-07-05T08:43:23.422911+00:00","updated_at":"2026-07-05T08:43:23.422911+00:00"}