{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:LDW3DIFRZ4IDFNRPR7UZMBEWWX","short_pith_number":"pith:LDW3DIFR","schema_version":"1.0","canonical_sha256":"58edb1a0b1cf1032b62f8fe9960496b5d138d2119e87a1daa4c957ed9531d266","source":{"kind":"arxiv","id":"2503.01739","version":2},"attestation_state":"computed","paper":{"title":"VideoUFO: A Million-Scale User-Focused Dataset for Text-to-Video Generation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Wenhao Wang, Yi Yang","submitted_at":"2025-03-03T17:00:36Z","abstract_excerpt":"Text-to-video generative models convert textual prompts into dynamic visual content, offering wide-ranging applications in film production, gaming, and education. However, their real-world performance often falls short of user expectations. One key reason is that these models have not been trained on videos related to some topics users want to create. In this paper, we propose VideoUFO, the first Video dataset specifically curated to align with Users' FOcus in real-world scenarios. Beyond this, our VideoUFO also features: (1) minimal (0.29%) overlap with existing video datasets, and (2) videos"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2503.01739","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2025-03-03T17:00:36Z","cross_cats_sorted":[],"title_canon_sha256":"686ffa6d15830e9afe8bbcd49c30354e512bb74204251c1885dae9d69096c92c","abstract_canon_sha256":"f7e305ab0ea8aeb9bb1396b490d6395542e690094693102a24a8041c60e7f2b4"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:02:18.822293Z","signature_b64":"D4FKOZ9mF2XWCZ6RJheD7ikDeaV54JKqauHL8EeAZaChRKDHj2mwqV+5yIXnLG28FppTa8mDDcmuE4zFDNt2Bg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"58edb1a0b1cf1032b62f8fe9960496b5d138d2119e87a1daa4c957ed9531d266","last_reissued_at":"2026-07-05T11:02:18.821787Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:02:18.821787Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"VideoUFO: A Million-Scale User-Focused Dataset for Text-to-Video Generation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Wenhao Wang, Yi Yang","submitted_at":"2025-03-03T17:00:36Z","abstract_excerpt":"Text-to-video generative models convert textual prompts into dynamic visual content, offering wide-ranging applications in film production, gaming, and education. However, their real-world performance often falls short of user expectations. One key reason is that these models have not been trained on videos related to some topics users want to create. In this paper, we propose VideoUFO, the first Video dataset specifically curated to align with Users' FOcus in real-world scenarios. Beyond this, our VideoUFO also features: (1) minimal (0.29%) overlap with existing video datasets, and (2) videos"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2503.01739","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2503.01739/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2503.01739","created_at":"2026-07-05T11:02:18.821849+00:00"},{"alias_kind":"arxiv_version","alias_value":"2503.01739v2","created_at":"2026-07-05T11:02:18.821849+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2503.01739","created_at":"2026-07-05T11:02:18.821849+00:00"},{"alias_kind":"pith_short_12","alias_value":"LDW3DIFRZ4ID","created_at":"2026-07-05T11:02:18.821849+00:00"},{"alias_kind":"pith_short_16","alias_value":"LDW3DIFRZ4IDFNRP","created_at":"2026-07-05T11:02:18.821849+00:00"},{"alias_kind":"pith_short_8","alias_value":"LDW3DIFR","created_at":"2026-07-05T11:02:18.821849+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.24937","citing_title":"The Hitchhiker's Guide to Agentic AI: From Foundations to Systems","ref_index":201,"is_internal_anchor":false},{"citing_arxiv_id":"2606.09639","citing_title":"CineDance: Towards Next-Generation Multi-Shot Long-Form Cinematic Audio-Video Generation","ref_index":70,"is_internal_anchor":false},{"citing_arxiv_id":"2607.00858","citing_title":"MoVA: Learning Asymmetric Dual Projections for Modular Long Video-Text Alignment","ref_index":53,"is_internal_anchor":false},{"citing_arxiv_id":"2601.04068","citing_title":"Mind the Generative Details: Direct Localized Detail Preference Optimization for Video Diffusion Models","ref_index":63,"is_internal_anchor":false},{"citing_arxiv_id":"2601.04068","citing_title":"Mind the Generative Details: Direct Localized Detail Preference Optimization for Video Diffusion Models","ref_index":63,"is_internal_anchor":false},{"citing_arxiv_id":"2604.11627","citing_title":"POINTS-Long: Adaptive Dual-Mode Visual Reasoning in MLLMs","ref_index":87,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/LDW3DIFRZ4IDFNRPR7UZMBEWWX","json":"https://pith.science/pith/LDW3DIFRZ4IDFNRPR7UZMBEWWX.json","graph_json":"https://pith.science/api/pith-number/LDW3DIFRZ4IDFNRPR7UZMBEWWX/graph.json","events_json":"https://pith.science/api/pith-number/LDW3DIFRZ4IDFNRPR7UZMBEWWX/events.json","paper":"https://pith.science/paper/LDW3DIFR"},"agent_actions":{"view_html":"https://pith.science/pith/LDW3DIFRZ4IDFNRPR7UZMBEWWX","download_json":"https://pith.science/pith/LDW3DIFRZ4IDFNRPR7UZMBEWWX.json","view_paper":"https://pith.science/paper/LDW3DIFR","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2503.01739&json=true","fetch_graph":"https://pith.science/api/pith-number/LDW3DIFRZ4IDFNRPR7UZMBEWWX/graph.json","fetch_events":"https://pith.science/api/pith-number/LDW3DIFRZ4IDFNRPR7UZMBEWWX/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/LDW3DIFRZ4IDFNRPR7UZMBEWWX/action/timestamp_anchor","attest_storage":"https://pith.science/pith/LDW3DIFRZ4IDFNRPR7UZMBEWWX/action/storage_attestation","attest_author":"https://pith.science/pith/LDW3DIFRZ4IDFNRPR7UZMBEWWX/action/author_attestation","sign_citation":"https://pith.science/pith/LDW3DIFRZ4IDFNRPR7UZMBEWWX/action/citation_signature","submit_replication":"https://pith.science/pith/LDW3DIFRZ4IDFNRPR7UZMBEWWX/action/replication_record"}},"created_at":"2026-07-05T11:02:18.821849+00:00","updated_at":"2026-07-05T11:02:18.821849+00:00"}