{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:QA55MJPSTOCNXGIG4OPUPOV5ZD","short_pith_number":"pith:QA55MJPS","schema_version":"1.0","canonical_sha256":"803bd625f29b84db9906e39f47babdc8d168b4ce85671a0fab296e9acd0d86d3","source":{"kind":"arxiv","id":"2410.02757","version":2},"attestation_state":"computed","paper":{"title":"Loong: Generating Minute-level Long Videos with Autoregressive Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Bingyi Kang, Daquan Zhou, Jiashi Feng, Tianwei Xiong, Xihui Liu, Yang Zhao, Yuqing Wang, Zhijie Lin","submitted_at":"2024-10-03T17:59:02Z","abstract_excerpt":"It is desirable but challenging to generate content-rich long videos in the scale of minutes. Autoregressive large language models (LLMs) have achieved great success in generating coherent and long sequences of tokens in the domain of natural language processing, while the exploration of autoregressive LLMs for video generation is limited to generating short videos of several seconds. In this work, we conduct a deep analysis of the challenges that prevent autoregressive LLM-based video generators from generating long videos. Based on the observations and analysis, we propose Loong, a new autor"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.02757","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2024-10-03T17:59:02Z","cross_cats_sorted":[],"title_canon_sha256":"39c21c830236ebd470db1f24af59fcb9c9dabad91f616a2fcf56ab2bc22dab92","abstract_canon_sha256":"06d2a08b7f2b866268e9f0fbaab74e80afebc8ed04157c67f3ad06c57e6bd9cd"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:43:01.823825Z","signature_b64":"XR4nomqR7wshaorvMIVtCB0GXux1PYyLAFw2oJflvDLj/3Rg3oaYCUTbOZSm7JsXoX37LciP6q+pB9p5VBovBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"803bd625f29b84db9906e39f47babdc8d168b4ce85671a0fab296e9acd0d86d3","last_reissued_at":"2026-07-05T10:43:01.823362Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:43:01.823362Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Loong: Generating Minute-level Long Videos with Autoregressive Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Bingyi Kang, Daquan Zhou, Jiashi Feng, Tianwei Xiong, Xihui Liu, Yang Zhao, Yuqing Wang, Zhijie Lin","submitted_at":"2024-10-03T17:59:02Z","abstract_excerpt":"It is desirable but challenging to generate content-rich long videos in the scale of minutes. Autoregressive large language models (LLMs) have achieved great success in generating coherent and long sequences of tokens in the domain of natural language processing, while the exploration of autoregressive LLMs for video generation is limited to generating short videos of several seconds. In this work, we conduct a deep analysis of the challenges that prevent autoregressive LLM-based video generators from generating long videos. Based on the observations and analysis, we propose Loong, a new autor"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.02757","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.02757/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.02757","created_at":"2026-07-05T10:43:01.823419+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.02757v2","created_at":"2026-07-05T10:43:01.823419+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.02757","created_at":"2026-07-05T10:43:01.823419+00:00"},{"alias_kind":"pith_short_12","alias_value":"QA55MJPSTOCN","created_at":"2026-07-05T10:43:01.823419+00:00"},{"alias_kind":"pith_short_16","alias_value":"QA55MJPSTOCNXGIG","created_at":"2026-07-05T10:43:01.823419+00:00"},{"alias_kind":"pith_short_8","alias_value":"QA55MJPS","created_at":"2026-07-05T10:43:01.823419+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":17,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.22370","citing_title":"Towards Error-Free Long Video Generation","ref_index":37,"is_internal_anchor":false},{"citing_arxiv_id":"2606.07508","citing_title":"Streaming Video Generation with Streaming Force Control","ref_index":57,"is_internal_anchor":false},{"citing_arxiv_id":"2605.31603","citing_title":"Lumos-Nexus: Efficient Frequency Bridging with Homogeneous Latent Space for Video Unified Models","ref_index":47,"is_internal_anchor":false},{"citing_arxiv_id":"2606.31326","citing_title":"Bridging Video Understanding and Generation in a Unified Framework","ref_index":62,"is_internal_anchor":false},{"citing_arxiv_id":"2605.30519","citing_title":"OmniMem: Scalable and Adaptive Memory Retrieval for Long Video Generation","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2603.03066","citing_title":"EduVQA: Towards Concept-Aware Assessment of Educational AI-Generated Videos","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2602.23058","citing_title":"GeoWorld: Geometric World Models","ref_index":78,"is_internal_anchor":false},{"citing_arxiv_id":"2510.20206","citing_title":"RAPO++: Cross-Stage Prompt Optimization for Text-to-Video Generation via Data Alignment and Test-Time Scaling","ref_index":81,"is_internal_anchor":false},{"citing_arxiv_id":"2412.14169","citing_title":"Autoregressive Video Generation without Vector Quantization","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2511.20714","citing_title":"Inferix: A Block-Diffusion based Next-Generation Inference Engine for World Simulation","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2509.25161","citing_title":"Rolling Forcing: Autoregressive Long Video Diffusion in Real Time","ref_index":97,"is_internal_anchor":false},{"citing_arxiv_id":"2602.07775","citing_title":"Rolling Sink: Bridging Limited-Horizon Training and Open-Ended Testing in Autoregressive Video Diffusion","ref_index":93,"is_internal_anchor":false},{"citing_arxiv_id":"2605.14487","citing_title":"Head Forcing: Long Autoregressive Video Generation via Head Heterogeneity","ref_index":52,"is_internal_anchor":false},{"citing_arxiv_id":"2603.28489","citing_title":"Video Generation Models as World Models: Efficient Paradigms, Architectures and Algorithms","ref_index":66,"is_internal_anchor":false},{"citing_arxiv_id":"2604.25819","citing_title":"Mutual Forcing: Dual-Mode Self-Evolution for Fast Autoregressive Audio-Video Character Generation","ref_index":50,"is_internal_anchor":false},{"citing_arxiv_id":"2506.08009","citing_title":"Self Forcing: Bridging the Train-Test Gap in Autoregressive Video Diffusion","ref_index":86,"is_internal_anchor":false},{"citing_arxiv_id":"2604.07209","citing_title":"INSPATIO-WORLD: A Real-Time 4D World Simulator via Spatiotemporal Autoregressive Modeling","ref_index":83,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/QA55MJPSTOCNXGIG4OPUPOV5ZD","json":"https://pith.science/pith/QA55MJPSTOCNXGIG4OPUPOV5ZD.json","graph_json":"https://pith.science/api/pith-number/QA55MJPSTOCNXGIG4OPUPOV5ZD/graph.json","events_json":"https://pith.science/api/pith-number/QA55MJPSTOCNXGIG4OPUPOV5ZD/events.json","paper":"https://pith.science/paper/QA55MJPS"},"agent_actions":{"view_html":"https://pith.science/pith/QA55MJPSTOCNXGIG4OPUPOV5ZD","download_json":"https://pith.science/pith/QA55MJPSTOCNXGIG4OPUPOV5ZD.json","view_paper":"https://pith.science/paper/QA55MJPS","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.02757&json=true","fetch_graph":"https://pith.science/api/pith-number/QA55MJPSTOCNXGIG4OPUPOV5ZD/graph.json","fetch_events":"https://pith.science/api/pith-number/QA55MJPSTOCNXGIG4OPUPOV5ZD/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/QA55MJPSTOCNXGIG4OPUPOV5ZD/action/timestamp_anchor","attest_storage":"https://pith.science/pith/QA55MJPSTOCNXGIG4OPUPOV5ZD/action/storage_attestation","attest_author":"https://pith.science/pith/QA55MJPSTOCNXGIG4OPUPOV5ZD/action/author_attestation","sign_citation":"https://pith.science/pith/QA55MJPSTOCNXGIG4OPUPOV5ZD/action/citation_signature","submit_replication":"https://pith.science/pith/QA55MJPSTOCNXGIG4OPUPOV5ZD/action/replication_record"}},"created_at":"2026-07-05T10:43:01.823419+00:00","updated_at":"2026-07-05T10:43:01.823419+00:00"}