{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:DIF3N3DR3WYICBAXHZMTV6XLMS","short_pith_number":"pith:DIF3N3DR","schema_version":"1.0","canonical_sha256":"1a0bb6ec71ddb08104173e593afaeb6481e61a0fa324840e325b014e0d621b54","source":{"kind":"arxiv","id":"2508.03694","version":1},"attestation_state":"computed","paper":{"title":"LongVie: Multimodal-Guided Controllable Ultra-Long Video Generation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Chenyang Si, Jianfeng Feng, Jianxiong Gao, Xian Liu, Yanwei Fu, Yu Qiao, Zhaoxi Chen, Ziwei Liu","submitted_at":"2025-08-05T17:59:58Z","abstract_excerpt":"Controllable ultra-long video generation is a fundamental yet challenging task. Although existing methods are effective for short clips, they struggle to scale due to issues such as temporal inconsistency and visual degradation. In this paper, we initially investigate and identify three key factors: separate noise initialization, independent control signal normalization, and the limitations of single-modality guidance. To address these issues, we propose LongVie, an end-to-end autoregressive framework for controllable long video generation. LongVie introduces two core designs to ensure tempora"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2508.03694","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2025-08-05T17:59:58Z","cross_cats_sorted":[],"title_canon_sha256":"33de3bef1786be3cdca04cf8b2bc78a7f3950b2b1ca77b4bdfa17f630d4e64d6","abstract_canon_sha256":"7bc011f9afddbadd13591b0ab99915ee901be10c763fef2a2e2d9f67fb72543f"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:49:01.384716Z","signature_b64":"66jKZ8aJltobtgq4sLPHgxYOMdD+esi+ljLnp+ar6kDSgGIem21gxxZgAHZ3TwIjP8DAGA/SwmMu7WyycwONDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"1a0bb6ec71ddb08104173e593afaeb6481e61a0fa324840e325b014e0d621b54","last_reissued_at":"2026-07-05T11:49:01.384267Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:49:01.384267Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"LongVie: Multimodal-Guided Controllable Ultra-Long Video Generation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Chenyang Si, Jianfeng Feng, Jianxiong Gao, Xian Liu, Yanwei Fu, Yu Qiao, Zhaoxi Chen, Ziwei Liu","submitted_at":"2025-08-05T17:59:58Z","abstract_excerpt":"Controllable ultra-long video generation is a fundamental yet challenging task. Although existing methods are effective for short clips, they struggle to scale due to issues such as temporal inconsistency and visual degradation. In this paper, we initially investigate and identify three key factors: separate noise initialization, independent control signal normalization, and the limitations of single-modality guidance. To address these issues, we propose LongVie, an end-to-end autoregressive framework for controllable long video generation. LongVie introduces two core designs to ensure tempora"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2508.03694","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2508.03694/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2508.03694","created_at":"2026-07-05T11:49:01.384323+00:00"},{"alias_kind":"arxiv_version","alias_value":"2508.03694v1","created_at":"2026-07-05T11:49:01.384323+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2508.03694","created_at":"2026-07-05T11:49:01.384323+00:00"},{"alias_kind":"pith_short_12","alias_value":"DIF3N3DR3WYI","created_at":"2026-07-05T11:49:01.384323+00:00"},{"alias_kind":"pith_short_16","alias_value":"DIF3N3DR3WYICBAX","created_at":"2026-07-05T11:49:01.384323+00:00"},{"alias_kind":"pith_short_8","alias_value":"DIF3N3DR","created_at":"2026-07-05T11:49:01.384323+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":7,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.30351","citing_title":"VideoMLA: Low-Rank Latent KV Cache for Minute-Scale Autoregressive Video Diffusion","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2605.23508","citing_title":"DrawVideo: Generating Long Video from Storyboard Keyframe Sketches","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2512.04678","citing_title":"Reward Forcing: Efficient Streaming Video Generation with Rewarded Distribution Matching Distillation","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2509.22622","citing_title":"LongLive: Real-time Interactive Long Video Generation","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2605.15199","citing_title":"EntityBench: Towards Entity-Consistent Long-Range Multi-Shot Video Generation","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2605.03849","citing_title":"Stream-R1: Reliability-Perplexity Aware Reward Distillation for Streaming Video Generation","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2604.06339","citing_title":"Evolution of Video Generative Foundations","ref_index":278,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/DIF3N3DR3WYICBAXHZMTV6XLMS","json":"https://pith.science/pith/DIF3N3DR3WYICBAXHZMTV6XLMS.json","graph_json":"https://pith.science/api/pith-number/DIF3N3DR3WYICBAXHZMTV6XLMS/graph.json","events_json":"https://pith.science/api/pith-number/DIF3N3DR3WYICBAXHZMTV6XLMS/events.json","paper":"https://pith.science/paper/DIF3N3DR"},"agent_actions":{"view_html":"https://pith.science/pith/DIF3N3DR3WYICBAXHZMTV6XLMS","download_json":"https://pith.science/pith/DIF3N3DR3WYICBAXHZMTV6XLMS.json","view_paper":"https://pith.science/paper/DIF3N3DR","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2508.03694&json=true","fetch_graph":"https://pith.science/api/pith-number/DIF3N3DR3WYICBAXHZMTV6XLMS/graph.json","fetch_events":"https://pith.science/api/pith-number/DIF3N3DR3WYICBAXHZMTV6XLMS/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/DIF3N3DR3WYICBAXHZMTV6XLMS/action/timestamp_anchor","attest_storage":"https://pith.science/pith/DIF3N3DR3WYICBAXHZMTV6XLMS/action/storage_attestation","attest_author":"https://pith.science/pith/DIF3N3DR3WYICBAXHZMTV6XLMS/action/author_attestation","sign_citation":"https://pith.science/pith/DIF3N3DR3WYICBAXHZMTV6XLMS/action/citation_signature","submit_replication":"https://pith.science/pith/DIF3N3DR3WYICBAXHZMTV6XLMS/action/replication_record"}},"created_at":"2026-07-05T11:49:01.384323+00:00","updated_at":"2026-07-05T11:49:01.384323+00:00"}