{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:KQ44SNTVLXWTHGKVAE6E44PNII","short_pith_number":"pith:KQ44SNTV","schema_version":"1.0","canonical_sha256":"5439c936755ded339955013c4e71ed423d430b006eedfbcf89c392f64e01940f","source":{"kind":"arxiv","id":"2406.18583","version":1},"attestation_state":"computed","paper":{"title":"Lumina-Next: Making Lumina-T2X Stronger and Faster with Next-DiT","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CV","authors_text":"Dingning Liu, Dongyang Liu, Fu-Yun Wang, Han Xiao, Hongsheng Li, Kaipeng Zhang, Le Zhuo, Lirui Zhao, Peng Gao, Rongjie Huang, Ruoyi Du, Si Liu, Wanli Ouyang, Wenze Liu, Xiangyang Zhu, Xiangyu Yue, Xu Luo, Yangguang Li, Yu Qiao, Zehan Wang, Zhanyu Ma, Ziwei Liu","submitted_at":"2024-06-05T17:53:26Z","abstract_excerpt":"Lumina-T2X is a nascent family of Flow-based Large Diffusion Transformers that establishes a unified framework for transforming noise into various modalities, such as images and videos, conditioned on text instructions. Despite its promising capabilities, Lumina-T2X still encounters challenges including training instability, slow inference, and extrapolation artifacts. In this paper, we present Lumina-Next, an improved version of Lumina-T2X, showcasing stronger generation performance with increased training and inference efficiency. We begin with a comprehensive analysis of the Flag-DiT archit"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2406.18583","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2024-06-05T17:53:26Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"885f1bfb3612b58396c807263e1eae8887a84d4ae9b056ea92ca246e9551eed5","abstract_canon_sha256":"5277d0d3f4199ade81e8c55db9fed01cc1a4fbf23fdff24bf1337b5d1c7d0c45"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:37:22.886369Z","signature_b64":"nHm0x6Ywl+1PY58F9qso2JV9TSnZgix7/3gZh1bzy6AuOrNK3/IwdzUCo7tvctB5sVkg0EkYQK+Xmg0b3dqJCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"5439c936755ded339955013c4e71ed423d430b006eedfbcf89c392f64e01940f","last_reissued_at":"2026-07-05T08:37:22.885887Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:37:22.885887Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Lumina-Next: Making Lumina-T2X Stronger and Faster with Next-DiT","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CV","authors_text":"Dingning Liu, Dongyang Liu, Fu-Yun Wang, Han Xiao, Hongsheng Li, Kaipeng Zhang, Le Zhuo, Lirui Zhao, Peng Gao, Rongjie Huang, Ruoyi Du, Si Liu, Wanli Ouyang, Wenze Liu, Xiangyang Zhu, Xiangyu Yue, Xu Luo, Yangguang Li, Yu Qiao, Zehan Wang, Zhanyu Ma, Ziwei Liu","submitted_at":"2024-06-05T17:53:26Z","abstract_excerpt":"Lumina-T2X is a nascent family of Flow-based Large Diffusion Transformers that establishes a unified framework for transforming noise into various modalities, such as images and videos, conditioned on text instructions. Despite its promising capabilities, Lumina-T2X still encounters challenges including training instability, slow inference, and extrapolation artifacts. In this paper, we present Lumina-Next, an improved version of Lumina-T2X, showcasing stronger generation performance with increased training and inference efficiency. We begin with a comprehensive analysis of the Flag-DiT archit"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.18583","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.18583/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2406.18583","created_at":"2026-07-05T08:37:22.885948+00:00"},{"alias_kind":"arxiv_version","alias_value":"2406.18583v1","created_at":"2026-07-05T08:37:22.885948+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.18583","created_at":"2026-07-05T08:37:22.885948+00:00"},{"alias_kind":"pith_short_12","alias_value":"KQ44SNTVLXWT","created_at":"2026-07-05T08:37:22.885948+00:00"},{"alias_kind":"pith_short_16","alias_value":"KQ44SNTVLXWTHGKV","created_at":"2026-07-05T08:37:22.885948+00:00"},{"alias_kind":"pith_short_8","alias_value":"KQ44SNTV","created_at":"2026-07-05T08:37:22.885948+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":14,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.05196","citing_title":"Unified Audio Intelligence Without Regressing on Text Intelligence","ref_index":181,"is_internal_anchor":true},{"citing_arxiv_id":"2605.28615","citing_title":"Compositional Text-to-Image Generation Via Region-aware Bimodal Direct Preference Optimization","ref_index":58,"is_internal_anchor":false},{"citing_arxiv_id":"2503.02537","citing_title":"RectifiedHR: Enable Efficient High-Resolution Synthesis via Energy Rectification","ref_index":62,"is_internal_anchor":false},{"citing_arxiv_id":"2605.21272","citing_title":"MONET: A Massive, Open, Non-redundant and Enriched Text-to-image dataset","ref_index":114,"is_internal_anchor":false},{"citing_arxiv_id":"2605.15684","citing_title":"ElasticDiT: Efficient Diffusion Transformers via Elastic Architecture and Sparse Attention for High-Resolution Image Generation on Mobile Devices","ref_index":36,"is_internal_anchor":false},{"citing_arxiv_id":"2506.01666","citing_title":"Synthesis of discrete-continuous quantum circuits with multimodal diffusion models","ref_index":70,"is_internal_anchor":false},{"citing_arxiv_id":"2412.14169","citing_title":"Autoregressive Video Generation without Vector Quantization","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2505.05472","citing_title":"Mogao: An Omni Foundation Model for Interleaved Multi-Modal Generation","ref_index":100,"is_internal_anchor":false},{"citing_arxiv_id":"2511.20645","citing_title":"PixelDiT: Pixel Diffusion Transformers for Image Generation","ref_index":46,"is_internal_anchor":false},{"citing_arxiv_id":"2410.10629","citing_title":"SANA: Efficient High-Resolution Image Synthesis with Linear Diffusion Transformers","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2504.06256","citing_title":"Transfer between Modalities with MetaQueries","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2505.09568","citing_title":"BLIP3-o: A Family of Fully Open Unified Multimodal Models-Architecture, Training and Dataset","ref_index":44,"is_internal_anchor":false},{"citing_arxiv_id":"2409.18869","citing_title":"Emu3: Next-Token Prediction is All You Need","ref_index":110,"is_internal_anchor":false},{"citing_arxiv_id":"2501.17811","citing_title":"Janus-Pro: Unified Multimodal Understanding and Generation with Data and Model Scaling","ref_index":57,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/KQ44SNTVLXWTHGKVAE6E44PNII","json":"https://pith.science/pith/KQ44SNTVLXWTHGKVAE6E44PNII.json","graph_json":"https://pith.science/api/pith-number/KQ44SNTVLXWTHGKVAE6E44PNII/graph.json","events_json":"https://pith.science/api/pith-number/KQ44SNTVLXWTHGKVAE6E44PNII/events.json","paper":"https://pith.science/paper/KQ44SNTV"},"agent_actions":{"view_html":"https://pith.science/pith/KQ44SNTVLXWTHGKVAE6E44PNII","download_json":"https://pith.science/pith/KQ44SNTVLXWTHGKVAE6E44PNII.json","view_paper":"https://pith.science/paper/KQ44SNTV","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2406.18583&json=true","fetch_graph":"https://pith.science/api/pith-number/KQ44SNTVLXWTHGKVAE6E44PNII/graph.json","fetch_events":"https://pith.science/api/pith-number/KQ44SNTVLXWTHGKVAE6E44PNII/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/KQ44SNTVLXWTHGKVAE6E44PNII/action/timestamp_anchor","attest_storage":"https://pith.science/pith/KQ44SNTVLXWTHGKVAE6E44PNII/action/storage_attestation","attest_author":"https://pith.science/pith/KQ44SNTVLXWTHGKVAE6E44PNII/action/author_attestation","sign_citation":"https://pith.science/pith/KQ44SNTVLXWTHGKVAE6E44PNII/action/citation_signature","submit_replication":"https://pith.science/pith/KQ44SNTVLXWTHGKVAE6E44PNII/action/replication_record"}},"created_at":"2026-07-05T08:37:22.885948+00:00","updated_at":"2026-07-05T08:37:22.885948+00:00"}