{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:OU26JZVXN7PFRPGWZIILPOBJGB","short_pith_number":"pith:OU26JZVX","schema_version":"1.0","canonical_sha256":"7535e4e6b76fde58bcd6ca10b7b829305595e3f26eab4bd522a9a21136384f9b","source":{"kind":"arxiv","id":"2104.14806","version":1},"attestation_state":"computed","paper":{"title":"GODIVA: Generating Open-DomaIn Videos from nAtural Descriptions","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Binyang Li, Chenfei Wu, Fan Yang, Guillermo Sapiro, Lei Ji, Lun Huang, Nan Duan, Qianxi Zhang","submitted_at":"2021-04-30T07:40:35Z","abstract_excerpt":"Generating videos from text is a challenging task due to its high computational requirements for training and infinite possible answers for evaluation. Existing works typically experiment on simple or small datasets, where the generalization ability is quite limited. In this work, we propose GODIVA, an open-domain text-to-video pretrained model that can generate videos from text in an auto-regressive manner using a three-dimensional sparse attention mechanism. We pretrain our model on Howto100M, a large-scale text-video dataset that contains more than 136 million text-video pairs. Experiments "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2104.14806","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2021-04-30T07:40:35Z","cross_cats_sorted":[],"title_canon_sha256":"b311166ff76651063503877899e661c1ebfbbffe9021a69a9183b1edbcec8023","abstract_canon_sha256":"01bcdca673ccfd100ddc1c22286c0de76c74f8341f266347bb340f720e036738"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T02:36:26.770578Z","signature_b64":"z8NLLASbP4ZRCMBYrStlko/gARdRG8EBMsfNQBW7zs95TB/MRqR5aNDnBEz/lttJTScOQycuGDtdOvfVmGf0Cw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"7535e4e6b76fde58bcd6ca10b7b829305595e3f26eab4bd522a9a21136384f9b","last_reissued_at":"2026-07-05T02:36:26.770113Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T02:36:26.770113Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"GODIVA: Generating Open-DomaIn Videos from nAtural Descriptions","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Binyang Li, Chenfei Wu, Fan Yang, Guillermo Sapiro, Lei Ji, Lun Huang, Nan Duan, Qianxi Zhang","submitted_at":"2021-04-30T07:40:35Z","abstract_excerpt":"Generating videos from text is a challenging task due to its high computational requirements for training and infinite possible answers for evaluation. Existing works typically experiment on simple or small datasets, where the generalization ability is quite limited. In this work, we propose GODIVA, an open-domain text-to-video pretrained model that can generate videos from text in an auto-regressive manner using a three-dimensional sparse attention mechanism. We pretrain our model on Howto100M, a large-scale text-video dataset that contains more than 136 million text-video pairs. Experiments "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2104.14806","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2104.14806/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2104.14806","created_at":"2026-07-05T02:36:26.770165+00:00"},{"alias_kind":"arxiv_version","alias_value":"2104.14806v1","created_at":"2026-07-05T02:36:26.770165+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2104.14806","created_at":"2026-07-05T02:36:26.770165+00:00"},{"alias_kind":"pith_short_12","alias_value":"OU26JZVXN7PF","created_at":"2026-07-05T02:36:26.770165+00:00"},{"alias_kind":"pith_short_16","alias_value":"OU26JZVXN7PFRPGW","created_at":"2026-07-05T02:36:26.770165+00:00"},{"alias_kind":"pith_short_8","alias_value":"OU26JZVX","created_at":"2026-07-05T02:36:26.770165+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":20,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.01556","citing_title":"TwinQuant: Learnable Subspace Decomposition for 4-Bit LLM Quantization","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2605.21466","citing_title":"StreamEdit: Training-Free Video Editing via Few-Step Streaming Video Generation","ref_index":79,"is_internal_anchor":false},{"citing_arxiv_id":"2605.15141","citing_title":"Causal Forcing++: Scalable Few-Step Autoregressive Diffusion Distillation for Real-Time Interactive Video Generation","ref_index":37,"is_internal_anchor":false},{"citing_arxiv_id":"2606.00793","citing_title":"MBench: A Comprehensive Benchmark on Memory Capability for Video World Models","ref_index":80,"is_internal_anchor":false},{"citing_arxiv_id":"2504.07940","citing_title":"Beyond the Frame: Generating 360 Panoramic Videos from Perspective Videos","ref_index":53,"is_internal_anchor":false},{"citing_arxiv_id":"2602.02214","citing_title":"Causal Forcing: Autoregressive Diffusion Distillation Done Right for High-Quality Real-Time Interactive Video Generation","ref_index":40,"is_internal_anchor":false},{"citing_arxiv_id":"2602.02214","citing_title":"Causal Forcing: Autoregressive Diffusion Distillation Done Right for High-Quality Real-Time Interactive Video Generation","ref_index":40,"is_internal_anchor":false},{"citing_arxiv_id":"2512.23994","citing_title":"PhyAVBench: A Challenging Audio Physics-Sensitivity Benchmark for Physically Grounded Text-to-Audio-Video Generation","ref_index":46,"is_internal_anchor":false},{"citing_arxiv_id":"2605.21466","citing_title":"StreamEdit: Training-Free Video Editing via Few-Step Streaming Video Generation","ref_index":79,"is_internal_anchor":false},{"citing_arxiv_id":"2605.16399","citing_title":"Stable and Near-Reversible Diffusion ODE Solvers for Image Editing","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2605.15661","citing_title":"VAGS: Velocity Adaptive Guidance Scale for Image Editing and Generation","ref_index":41,"is_internal_anchor":false},{"citing_arxiv_id":"2308.08089","citing_title":"DragNUWA: Fine-grained Control in Video Generation by Integrating Text, Image, and Trajectory","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2210.02399","citing_title":"Phenaki: Variable Length Video Generation From Open Domain Textual Description","ref_index":52,"is_internal_anchor":false},{"citing_arxiv_id":"2512.23994","citing_title":"PhyAVBench: A Challenging Audio Physics-Sensitivity Benchmark for Physically Grounded Text-to-Audio-Video Generation","ref_index":46,"is_internal_anchor":false},{"citing_arxiv_id":"2310.06114","citing_title":"Learning Interactive Real-World Simulators","ref_index":174,"is_internal_anchor":false},{"citing_arxiv_id":"2308.06571","citing_title":"ModelScope Text-to-Video Technical Report","ref_index":61,"is_internal_anchor":false},{"citing_arxiv_id":"2205.15868","citing_title":"CogVideo: Large-scale Pretraining for Text-to-Video Generation via Transformers","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2311.15127","citing_title":"Stable Video Diffusion: Scaling Latent Video Diffusion Models to Large Datasets","ref_index":103,"is_internal_anchor":false},{"citing_arxiv_id":"2604.20258","citing_title":"Rethinking Where to Edit: Task-Aware Localization for Instruction-Based Image Editing","ref_index":36,"is_internal_anchor":false},{"citing_arxiv_id":"2605.02417","citing_title":"DirectEdit: Step-Level Accurate Inversion for Flow-Based Image Editing","ref_index":34,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/OU26JZVXN7PFRPGWZIILPOBJGB","json":"https://pith.science/pith/OU26JZVXN7PFRPGWZIILPOBJGB.json","graph_json":"https://pith.science/api/pith-number/OU26JZVXN7PFRPGWZIILPOBJGB/graph.json","events_json":"https://pith.science/api/pith-number/OU26JZVXN7PFRPGWZIILPOBJGB/events.json","paper":"https://pith.science/paper/OU26JZVX"},"agent_actions":{"view_html":"https://pith.science/pith/OU26JZVXN7PFRPGWZIILPOBJGB","download_json":"https://pith.science/pith/OU26JZVXN7PFRPGWZIILPOBJGB.json","view_paper":"https://pith.science/paper/OU26JZVX","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2104.14806&json=true","fetch_graph":"https://pith.science/api/pith-number/OU26JZVXN7PFRPGWZIILPOBJGB/graph.json","fetch_events":"https://pith.science/api/pith-number/OU26JZVXN7PFRPGWZIILPOBJGB/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/OU26JZVXN7PFRPGWZIILPOBJGB/action/timestamp_anchor","attest_storage":"https://pith.science/pith/OU26JZVXN7PFRPGWZIILPOBJGB/action/storage_attestation","attest_author":"https://pith.science/pith/OU26JZVXN7PFRPGWZIILPOBJGB/action/author_attestation","sign_citation":"https://pith.science/pith/OU26JZVXN7PFRPGWZIILPOBJGB/action/citation_signature","submit_replication":"https://pith.science/pith/OU26JZVXN7PFRPGWZIILPOBJGB/action/replication_record"}},"created_at":"2026-07-05T02:36:26.770165+00:00","updated_at":"2026-07-05T02:36:26.770165+00:00"}