{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:QDBOGXFK462YR4NNZXFNVGU3U2","short_pith_number":"pith:QDBOGXFK","schema_version":"1.0","canonical_sha256":"80c2e35caae7b588f1adcdcada9a9ba6a5183f7ceccc2c327538f9714d26a1f2","source":{"kind":"arxiv","id":"2405.04233","version":1},"attestation_state":"computed","paper":{"title":"Vidu: a Highly Consistent, Dynamic and Skilled Text-to-Video Generator with Diffusion Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CV","authors_text":"Chendong Xiang, Fan Bao, Gang Yue, Guande He, Hongzhou Zhu, Jun Zhu, Kaiwen Zheng, Min Zhao, Shilong Liu, Yaole Wang","submitted_at":"2024-05-07T11:52:49Z","abstract_excerpt":"We introduce Vidu, a high-performance text-to-video generator that is capable of producing 1080p videos up to 16 seconds in a single generation. Vidu is a diffusion model with U-ViT as its backbone, which unlocks the scalability and the capability for handling long videos. Vidu exhibits strong coherence and dynamism, and is capable of generating both realistic and imaginative videos, as well as understanding some professional photography techniques, on par with Sora -- the most powerful reported text-to-video generator. Finally, we perform initial experiments on other controllable video genera"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2405.04233","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-05-07T11:52:49Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"d2093427949eb87fa2a32df127f2ade39f74b9d5d0d2b7a86779e6e27eb36769","abstract_canon_sha256":"b4992e326f5ef18281323cac0a3248f6fac62e884d02dc92657bbe32a2778ba1"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:16:33.777924Z","signature_b64":"u3ymjd8k/lQ+S1Dk6y2YSXEh3Fy5olHzf54DQaFBq7e+wwkFeHaH3z7/jTmrStzYLW/0K2vyD1yYKoP/e/HvBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"80c2e35caae7b588f1adcdcada9a9ba6a5183f7ceccc2c327538f9714d26a1f2","last_reissued_at":"2026-07-05T08:16:33.777454Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:16:33.777454Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Vidu: a Highly Consistent, Dynamic and Skilled Text-to-Video Generator with Diffusion Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CV","authors_text":"Chendong Xiang, Fan Bao, Gang Yue, Guande He, Hongzhou Zhu, Jun Zhu, Kaiwen Zheng, Min Zhao, Shilong Liu, Yaole Wang","submitted_at":"2024-05-07T11:52:49Z","abstract_excerpt":"We introduce Vidu, a high-performance text-to-video generator that is capable of producing 1080p videos up to 16 seconds in a single generation. Vidu is a diffusion model with U-ViT as its backbone, which unlocks the scalability and the capability for handling long videos. Vidu exhibits strong coherence and dynamism, and is capable of generating both realistic and imaginative videos, as well as understanding some professional photography techniques, on par with Sora -- the most powerful reported text-to-video generator. Finally, we perform initial experiments on other controllable video genera"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2405.04233","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2405.04233/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2405.04233","created_at":"2026-07-05T08:16:33.777513+00:00"},{"alias_kind":"arxiv_version","alias_value":"2405.04233v1","created_at":"2026-07-05T08:16:33.777513+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2405.04233","created_at":"2026-07-05T08:16:33.777513+00:00"},{"alias_kind":"pith_short_12","alias_value":"QDBOGXFK462Y","created_at":"2026-07-05T08:16:33.777513+00:00"},{"alias_kind":"pith_short_16","alias_value":"QDBOGXFK462YR4NN","created_at":"2026-07-05T08:16:33.777513+00:00"},{"alias_kind":"pith_short_8","alias_value":"QDBOGXFK","created_at":"2026-07-05T08:16:33.777513+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":37,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.25473","citing_title":"Causal-rCM: A Unified Teacher-Forcing and Self-Forcing Open Recipe for Autoregressive Diffusion Distillation in Streaming Video Generation and Interactive World Models","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2606.25306","citing_title":"Physics Question Scene Graph: Fine-grained Evaluation of Physical Plausibility in Text-to-Video Generation","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2606.20799","citing_title":"GroundShot: Visually Consistent Multi-Shot Long Video Generation via Entity-Grounded Shot Scheduling","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2606.19163","citing_title":"Pulse: Training Acceleration for Large Diffusion Models with Automatic Pipeline Parallelism","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2606.11670","citing_title":"ARGUS: Stacked Multi-View Identity Mosaic Injection for Subject-Preserving Video Generation","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2606.02402","citing_title":"Explainable Forensics of Manipulated Segments in Untrimmed Long Videos","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2606.31734","citing_title":"MemLearner: Learning to Query Context memory for Video World Models","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2605.14382","citing_title":"Delta Forcing: Trust Region Steering for Interactive Autoregressive Video Generation","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2605.15141","citing_title":"Causal Forcing++: Scalable Few-Step Autoregressive Diffusion Distillation for Real-Time Interactive Video Generation","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2605.30263","citing_title":"minWM: A Full-Stack Open-Source Framework for Real-Time Interactive Video World Models","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2606.01481","citing_title":"SafeGen-Bench: Benchmarking Safety in Image-Conditioned Text-to-Video Generation","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2503.20314","citing_title":"Wan: Open and Advanced Large-Scale Video Generative Models","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2602.02214","citing_title":"Causal Forcing: Autoregressive Diffusion Distillation Done Right for High-Quality Real-Time Interactive Video Generation","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2602.02214","citing_title":"Causal Forcing: Autoregressive Diffusion Distillation Done Right for High-Quality Real-Time Interactive Video Generation","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2604.27505","citing_title":"Leveraging Verifier-Based Reinforcement Learning in Image Editing","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2605.14382","citing_title":"Delta Forcing: Trust Region Steering for Interactive Autoregressive Video Generation","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2605.14382","citing_title":"Delta Forcing: Trust Region Steering for Interactive Autoregressive Video Generation","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17248","citing_title":"Image-to-Video Diffusion: From Foundations to Open Frontiers","ref_index":150,"is_internal_anchor":false},{"citing_arxiv_id":"2507.12768","citing_title":"AnyPos: Automated Task-Agnostic Actions for Bimanual Manipulation","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2406.03736","citing_title":"Your Absorbing Discrete Diffusion Secretly Models the Conditional Distributions of Clean Data","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2510.08431","citing_title":"Large Scale Diffusion Distillation via Score-Regularized Continuous-Time Consistency","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2507.12898","citing_title":"Vidar: Embodied Video Diffusion Model for Generalist Manipulation","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2602.13669","citing_title":"EchoTorrent: Towards Swift, Sustained, and Streaming Multi-Modal Video Generation","ref_index":53,"is_internal_anchor":false},{"citing_arxiv_id":"2605.14382","citing_title":"Delta Forcing: Trust Region Steering for Interactive Autoregressive Video Generation","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2604.03819","citing_title":"ActivityForensics: A Comprehensive Benchmark for Localizing Manipulated Activity in Videos","ref_index":1,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/QDBOGXFK462YR4NNZXFNVGU3U2","json":"https://pith.science/pith/QDBOGXFK462YR4NNZXFNVGU3U2.json","graph_json":"https://pith.science/api/pith-number/QDBOGXFK462YR4NNZXFNVGU3U2/graph.json","events_json":"https://pith.science/api/pith-number/QDBOGXFK462YR4NNZXFNVGU3U2/events.json","paper":"https://pith.science/paper/QDBOGXFK"},"agent_actions":{"view_html":"https://pith.science/pith/QDBOGXFK462YR4NNZXFNVGU3U2","download_json":"https://pith.science/pith/QDBOGXFK462YR4NNZXFNVGU3U2.json","view_paper":"https://pith.science/paper/QDBOGXFK","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2405.04233&json=true","fetch_graph":"https://pith.science/api/pith-number/QDBOGXFK462YR4NNZXFNVGU3U2/graph.json","fetch_events":"https://pith.science/api/pith-number/QDBOGXFK462YR4NNZXFNVGU3U2/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/QDBOGXFK462YR4NNZXFNVGU3U2/action/timestamp_anchor","attest_storage":"https://pith.science/pith/QDBOGXFK462YR4NNZXFNVGU3U2/action/storage_attestation","attest_author":"https://pith.science/pith/QDBOGXFK462YR4NNZXFNVGU3U2/action/author_attestation","sign_citation":"https://pith.science/pith/QDBOGXFK462YR4NNZXFNVGU3U2/action/citation_signature","submit_replication":"https://pith.science/pith/QDBOGXFK462YR4NNZXFNVGU3U2/action/replication_record"}},"created_at":"2026-07-05T08:16:33.777513+00:00","updated_at":"2026-07-05T08:16:33.777513+00:00"}