{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:4KKY6MQH6AXMNB6ENNJXU62Q2O","short_pith_number":"pith:4KKY6MQH","schema_version":"1.0","canonical_sha256":"e2958f3207f02ec687c46b537a7b50d3a4e8c9e8f33b97b2ba504d63cde9e48d","source":{"kind":"arxiv","id":"2309.00398","version":2},"attestation_state":"computed","paper":{"title":"VideoGen: A Reference-Guided Latent Diffusion Approach for High Definition Text-to-Video Generation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.MM"],"primary_cat":"cs.CV","authors_text":"Errui Ding, Fanglong Liu, Fu Li, Haocheng Feng, Jingdong Wang, Qi Zhang, Weihang Yuan, Wenqing Chu, Xin Li, Ye Wu","submitted_at":"2023-09-01T11:14:43Z","abstract_excerpt":"In this paper, we present VideoGen, a text-to-video generation approach, which can generate a high-definition video with high frame fidelity and strong temporal consistency using reference-guided latent diffusion. We leverage an off-the-shelf text-to-image generation model, e.g., Stable Diffusion, to generate an image with high content quality from the text prompt, as a reference image to guide video generation. Then, we introduce an efficient cascaded latent diffusion module conditioned on both the reference image and the text prompt, for generating latent video representations, followed by a"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2309.00398","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2023-09-01T11:14:43Z","cross_cats_sorted":["cs.MM"],"title_canon_sha256":"4650196613b3efc7162c69be4feee2ddbdb4c833ed102027984c5aa6c2048d33","abstract_canon_sha256":"1c55dae2dc87733a7c6475e74078579894f8ab41ef2e28b735c7e430e1cb408e"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:48:36.883270Z","signature_b64":"QdYPCJukGTZhVoosznrji4erLyiVcLM9xfZsZ73RJNTPv+sKMyUoPeg829mXck5+o6I93CTv41VFjXeXHuVLBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"e2958f3207f02ec687c46b537a7b50d3a4e8c9e8f33b97b2ba504d63cde9e48d","last_reissued_at":"2026-07-05T06:48:36.882799Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:48:36.882799Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"VideoGen: A Reference-Guided Latent Diffusion Approach for High Definition Text-to-Video Generation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.MM"],"primary_cat":"cs.CV","authors_text":"Errui Ding, Fanglong Liu, Fu Li, Haocheng Feng, Jingdong Wang, Qi Zhang, Weihang Yuan, Wenqing Chu, Xin Li, Ye Wu","submitted_at":"2023-09-01T11:14:43Z","abstract_excerpt":"In this paper, we present VideoGen, a text-to-video generation approach, which can generate a high-definition video with high frame fidelity and strong temporal consistency using reference-guided latent diffusion. We leverage an off-the-shelf text-to-image generation model, e.g., Stable Diffusion, to generate an image with high content quality from the text prompt, as a reference image to guide video generation. Then, we introduce an efficient cascaded latent diffusion module conditioned on both the reference image and the text prompt, for generating latent video representations, followed by a"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2309.00398","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2309.00398/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2309.00398","created_at":"2026-07-05T06:48:36.882848+00:00"},{"alias_kind":"arxiv_version","alias_value":"2309.00398v2","created_at":"2026-07-05T06:48:36.882848+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2309.00398","created_at":"2026-07-05T06:48:36.882848+00:00"},{"alias_kind":"pith_short_12","alias_value":"4KKY6MQH6AXM","created_at":"2026-07-05T06:48:36.882848+00:00"},{"alias_kind":"pith_short_16","alias_value":"4KKY6MQH6AXMNB6E","created_at":"2026-07-05T06:48:36.882848+00:00"},{"alias_kind":"pith_short_8","alias_value":"4KKY6MQH","created_at":"2026-07-05T06:48:36.882848+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.24107","citing_title":"DramaDirector: Geometry-Guided Short Drama Generation","ref_index":38,"is_internal_anchor":false},{"citing_arxiv_id":"2603.09283","citing_title":"From Ideal to Real: Stable Video Object Removal under Imperfect Conditions","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2310.19512","citing_title":"VideoCrafter1: Open Diffusion Models for High-Quality Video Generation","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11363","citing_title":"PresentAgent-2: Towards Generalist Multimodal Presentation Agents","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2410.13720","citing_title":"Movie Gen: A Cast of Media Foundation Models","ref_index":41,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/4KKY6MQH6AXMNB6ENNJXU62Q2O","json":"https://pith.science/pith/4KKY6MQH6AXMNB6ENNJXU62Q2O.json","graph_json":"https://pith.science/api/pith-number/4KKY6MQH6AXMNB6ENNJXU62Q2O/graph.json","events_json":"https://pith.science/api/pith-number/4KKY6MQH6AXMNB6ENNJXU62Q2O/events.json","paper":"https://pith.science/paper/4KKY6MQH"},"agent_actions":{"view_html":"https://pith.science/pith/4KKY6MQH6AXMNB6ENNJXU62Q2O","download_json":"https://pith.science/pith/4KKY6MQH6AXMNB6ENNJXU62Q2O.json","view_paper":"https://pith.science/paper/4KKY6MQH","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2309.00398&json=true","fetch_graph":"https://pith.science/api/pith-number/4KKY6MQH6AXMNB6ENNJXU62Q2O/graph.json","fetch_events":"https://pith.science/api/pith-number/4KKY6MQH6AXMNB6ENNJXU62Q2O/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/4KKY6MQH6AXMNB6ENNJXU62Q2O/action/timestamp_anchor","attest_storage":"https://pith.science/pith/4KKY6MQH6AXMNB6ENNJXU62Q2O/action/storage_attestation","attest_author":"https://pith.science/pith/4KKY6MQH6AXMNB6ENNJXU62Q2O/action/author_attestation","sign_citation":"https://pith.science/pith/4KKY6MQH6AXMNB6ENNJXU62Q2O/action/citation_signature","submit_replication":"https://pith.science/pith/4KKY6MQH6AXMNB6ENNJXU62Q2O/action/replication_record"}},"created_at":"2026-07-05T06:48:36.882848+00:00","updated_at":"2026-07-05T06:48:36.882848+00:00"}