{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:UMSMTM6AAZVLNVF24L45BJL6D4","short_pith_number":"pith:UMSMTM6A","schema_version":"1.0","canonical_sha256":"a324c9b3c0066ab6d4bae2f9d0a57e1f1eb99c4dc8ab9ca1d5365503d94ffbf3","source":{"kind":"arxiv","id":"2401.09414","version":1},"attestation_state":"computed","paper":{"title":"Vlogger: Make Your Dream A Vlog","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG","cs.MM"],"primary_cat":"cs.CV","authors_text":"Kunchang Li, Shaobin Zhuang, Xinyuan Chen, Yali Wang, Yaohui Wang, Yu Qiao, Ziwei Liu","submitted_at":"2024-01-17T18:55:12Z","abstract_excerpt":"In this work, we present Vlogger, a generic AI system for generating a minute-level video blog (i.e., vlog) of user descriptions. Different from short videos with a few seconds, vlog often contains a complex storyline with diversified scenes, which is challenging for most existing video generation approaches. To break through this bottleneck, our Vlogger smartly leverages Large Language Model (LLM) as Director and decomposes a long video generation task of vlog into four key stages, where we invoke various foundation models to play the critical roles of vlog professionals, including (1) Script"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2401.09414","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2024-01-17T18:55:12Z","cross_cats_sorted":["cs.AI","cs.LG","cs.MM"],"title_canon_sha256":"d84cf217ff269d1c8f57471cfbcbcc877d79649b716f4f3dadee3e4d62a3c18b","abstract_canon_sha256":"6ddb79b9c1efa1b2673e45e91a45b5f0c665602fcfca6b833615c4f99a797ef3"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:34:43.585082Z","signature_b64":"RYiidBfaeAspOn6W7BMKcbJkejutGWw7rMt538oR5nnk8eOKmx7RmjBve1ENBNMfBFTZbZ6N9pQjAjrNX6SmBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"a324c9b3c0066ab6d4bae2f9d0a57e1f1eb99c4dc8ab9ca1d5365503d94ffbf3","last_reissued_at":"2026-07-05T07:34:43.584671Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:34:43.584671Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Vlogger: Make Your Dream A Vlog","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG","cs.MM"],"primary_cat":"cs.CV","authors_text":"Kunchang Li, Shaobin Zhuang, Xinyuan Chen, Yali Wang, Yaohui Wang, Yu Qiao, Ziwei Liu","submitted_at":"2024-01-17T18:55:12Z","abstract_excerpt":"In this work, we present Vlogger, a generic AI system for generating a minute-level video blog (i.e., vlog) of user descriptions. Different from short videos with a few seconds, vlog often contains a complex storyline with diversified scenes, which is challenging for most existing video generation approaches. To break through this bottleneck, our Vlogger smartly leverages Large Language Model (LLM) as Director and decomposes a long video generation task of vlog into four key stages, where we invoke various foundation models to play the critical roles of vlog professionals, including (1) Script"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2401.09414","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2401.09414/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2401.09414","created_at":"2026-07-05T07:34:43.584726+00:00"},{"alias_kind":"arxiv_version","alias_value":"2401.09414v1","created_at":"2026-07-05T07:34:43.584726+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2401.09414","created_at":"2026-07-05T07:34:43.584726+00:00"},{"alias_kind":"pith_short_12","alias_value":"UMSMTM6AAZVL","created_at":"2026-07-05T07:34:43.584726+00:00"},{"alias_kind":"pith_short_16","alias_value":"UMSMTM6AAZVLNVF2","created_at":"2026-07-05T07:34:43.584726+00:00"},{"alias_kind":"pith_short_8","alias_value":"UMSMTM6A","created_at":"2026-07-05T07:34:43.584726+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.20799","citing_title":"GroundShot: Visually Consistent Multi-Shot Long Video Generation via Entity-Grounded Shot Scheduling","ref_index":66,"is_internal_anchor":false},{"citing_arxiv_id":"2402.17177","citing_title":"Sora: A Review on Background, Technology, Limitations, and Opportunities of Large Vision Models","ref_index":146,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/UMSMTM6AAZVLNVF24L45BJL6D4","json":"https://pith.science/pith/UMSMTM6AAZVLNVF24L45BJL6D4.json","graph_json":"https://pith.science/api/pith-number/UMSMTM6AAZVLNVF24L45BJL6D4/graph.json","events_json":"https://pith.science/api/pith-number/UMSMTM6AAZVLNVF24L45BJL6D4/events.json","paper":"https://pith.science/paper/UMSMTM6A"},"agent_actions":{"view_html":"https://pith.science/pith/UMSMTM6AAZVLNVF24L45BJL6D4","download_json":"https://pith.science/pith/UMSMTM6AAZVLNVF24L45BJL6D4.json","view_paper":"https://pith.science/paper/UMSMTM6A","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2401.09414&json=true","fetch_graph":"https://pith.science/api/pith-number/UMSMTM6AAZVLNVF24L45BJL6D4/graph.json","fetch_events":"https://pith.science/api/pith-number/UMSMTM6AAZVLNVF24L45BJL6D4/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/UMSMTM6AAZVLNVF24L45BJL6D4/action/timestamp_anchor","attest_storage":"https://pith.science/pith/UMSMTM6AAZVLNVF24L45BJL6D4/action/storage_attestation","attest_author":"https://pith.science/pith/UMSMTM6AAZVLNVF24L45BJL6D4/action/author_attestation","sign_citation":"https://pith.science/pith/UMSMTM6AAZVLNVF24L45BJL6D4/action/citation_signature","submit_replication":"https://pith.science/pith/UMSMTM6AAZVLNVF24L45BJL6D4/action/replication_record"}},"created_at":"2026-07-05T07:34:43.584726+00:00","updated_at":"2026-07-05T07:34:43.584726+00:00"}