{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:I4LLHNSONUBE7UWDRNLXMES73F","short_pith_number":"pith:I4LLHNSO","schema_version":"1.0","canonical_sha256":"4716b3b64e6d024fd2c38b5776125fd94d44ec2179112bea251ed20094c9399c","source":{"kind":"arxiv","id":"2508.10711","version":2},"attestation_state":"computed","paper":{"title":"NextStep-1: Toward Autoregressive Image Generation with Continuous Tokens at Scale","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Ailin Huang, Bin Wang, Binxing Jiao, Changxin Miao, Daxin Jiang, Deshan Sun, Deyu Zhou, En Yu, Fukun Yin, Gang Yu, Guopeng Li, Hanpeng Hu, Haomiao Tang, Hao Nie, Haoran Lv, Hongyu Zhou, Jianjian Sun, Jian Zhou, Jia Wang, Jingwei Wu, Kaijun Tan, Kang An, Kangheng Lin, KenKun Liu, Liang Zhao, Mei Chen, NextStep Team: Chunrui Han, Peng Xing, Quan Sun, Rui Wang, Shiyu Liu, Shutao Xia, Tianhao You, Wei Ji, Xianfang Zeng, Xiangyu Zhang, Xin Han, Xuelin Zhang, Yana Wei, Yan Cai, Yanming Xu, Yibo Zhu, Yimin Jiang, Yingming Wang, Yuang Peng, Yucheng Han, Yu Zhou, Zheng Ge, Ziyang Meng","submitted_at":"2025-08-14T14:54:22Z","abstract_excerpt":"Prevailing autoregressive (AR) models for text-to-image generation either rely on heavy, computationally-intensive diffusion models to process continuous image tokens, or employ vector quantization (VQ) to obtain discrete tokens with quantization loss. In this paper, we push the autoregressive paradigm forward with NextStep-1, a 14B autoregressive model paired with a 157M flow matching head, training on discrete text tokens and continuous image tokens with next-token prediction objectives. NextStep-1 achieves state-of-the-art performance for autoregressive models in text-to-image generation ta"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2508.10711","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2025-08-14T14:54:22Z","cross_cats_sorted":[],"title_canon_sha256":"c11f0b356ba0b8114c16c7accd68c884618bd94447320c26233f4c7dd63a9a23","abstract_canon_sha256":"34e3e90487c63349bf2408879fe0b45235de1c87b2c75f81cdb44bdb1a4b3c24"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:55:22.299929Z","signature_b64":"kf7WUXIgXOebXFghk1I7gBhyFAlR7csF40r8B/L1BmE2OQJU54UXnean+x/ieipuMPVvK2wgnSWXrGNB1RJ/DQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"4716b3b64e6d024fd2c38b5776125fd94d44ec2179112bea251ed20094c9399c","last_reissued_at":"2026-07-05T11:55:22.299429Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:55:22.299429Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"NextStep-1: Toward Autoregressive Image Generation with Continuous Tokens at Scale","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Ailin Huang, Bin Wang, Binxing Jiao, Changxin Miao, Daxin Jiang, Deshan Sun, Deyu Zhou, En Yu, Fukun Yin, Gang Yu, Guopeng Li, Hanpeng Hu, Haomiao Tang, Hao Nie, Haoran Lv, Hongyu Zhou, Jianjian Sun, Jian Zhou, Jia Wang, Jingwei Wu, Kaijun Tan, Kang An, Kangheng Lin, KenKun Liu, Liang Zhao, Mei Chen, NextStep Team: Chunrui Han, Peng Xing, Quan Sun, Rui Wang, Shiyu Liu, Shutao Xia, Tianhao You, Wei Ji, Xianfang Zeng, Xiangyu Zhang, Xin Han, Xuelin Zhang, Yana Wei, Yan Cai, Yanming Xu, Yibo Zhu, Yimin Jiang, Yingming Wang, Yuang Peng, Yucheng Han, Yu Zhou, Zheng Ge, Ziyang Meng","submitted_at":"2025-08-14T14:54:22Z","abstract_excerpt":"Prevailing autoregressive (AR) models for text-to-image generation either rely on heavy, computationally-intensive diffusion models to process continuous image tokens, or employ vector quantization (VQ) to obtain discrete tokens with quantization loss. In this paper, we push the autoregressive paradigm forward with NextStep-1, a 14B autoregressive model paired with a 157M flow matching head, training on discrete text tokens and continuous image tokens with next-token prediction objectives. NextStep-1 achieves state-of-the-art performance for autoregressive models in text-to-image generation ta"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2508.10711","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2508.10711/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2508.10711","created_at":"2026-07-05T11:55:22.299490+00:00"},{"alias_kind":"arxiv_version","alias_value":"2508.10711v2","created_at":"2026-07-05T11:55:22.299490+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2508.10711","created_at":"2026-07-05T11:55:22.299490+00:00"},{"alias_kind":"pith_short_12","alias_value":"I4LLHNSONUBE","created_at":"2026-07-05T11:55:22.299490+00:00"},{"alias_kind":"pith_short_16","alias_value":"I4LLHNSONUBE7UWD","created_at":"2026-07-05T11:55:22.299490+00:00"},{"alias_kind":"pith_short_8","alias_value":"I4LLHNSO","created_at":"2026-07-05T11:55:22.299490+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":19,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.19531","citing_title":"ImageWAM: Do World Action Models Really Need Video Generation, or Just Image Editing?","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2606.06357","citing_title":"F3-Tokenizer: Taming Audio Autoencoder Latents for Understanding and Generation","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2606.31278","citing_title":"Editing Everything Everywhere All at Once","ref_index":94,"is_internal_anchor":false},{"citing_arxiv_id":"2605.25328","citing_title":"DIVA: Harnessing the Representation Divergence in Unified Multimodal Models for Mutual Reinforcement","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2605.26089","citing_title":"Channel-wise Vector Quantization","ref_index":36,"is_internal_anchor":false},{"citing_arxiv_id":"2605.21487","citing_title":"Uni-Edit: Intelligent Editing Is A General Task For Unified Model Tuning","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2605.21487","citing_title":"Uni-Edit: Intelligent Editing Is A General Task For Unified Model Tuning","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2602.06663","citing_title":"PlanViz: Evaluating Planning-Oriented Image Generation and Editing for Computer-Use Tasks","ref_index":38,"is_internal_anchor":false},{"citing_arxiv_id":"2602.12370","citing_title":"LLaMo: Scaling Pretrained Language Models for Unified Motion Understanding and Generation with Continuous Autoregressive Tokens","ref_index":62,"is_internal_anchor":false},{"citing_arxiv_id":"2604.02355","citing_title":"From Broad Exploration to Stable Synthesis: Entropy-Guided Optimization for Autoregressive Image Generation","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2603.28049","citing_title":"Drift-AR: Single-Step Visual Autoregressive Generation via Anti-Symmetric Drifting","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09430","citing_title":"FlashAR: Efficient Post-Training Acceleration for Autoregressive Image Generation","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10600","citing_title":"Generate \"Normal\", Edit Poisoned: Branding Injection via Hint Embedding in Image Editing","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09430","citing_title":"FlashAR: Efficient Post-Training Acceleration for Autoregressive Image Generation","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2604.24885","citing_title":"VibeToken: Scaling 1D Image Tokenizers and Autoregressive Models for Dynamic Resolution Generations","ref_index":38,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05781","citing_title":"Steering Visual Generation in Unified Multimodal Models with Understanding Supervision","ref_index":47,"is_internal_anchor":false},{"citing_arxiv_id":"2604.13030","citing_title":"Generative Refinement Networks for Visual Synthesis","ref_index":51,"is_internal_anchor":false},{"citing_arxiv_id":"2604.06966","citing_title":"MAR-GRPO: Stabilized GRPO for AR-diffusion Hybrid Image Generation","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2604.08121","citing_title":"Uni-ViGU: Towards Unified Video Generation and Understanding via A Diffusion-Based Video Generator","ref_index":12,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/I4LLHNSONUBE7UWDRNLXMES73F","json":"https://pith.science/pith/I4LLHNSONUBE7UWDRNLXMES73F.json","graph_json":"https://pith.science/api/pith-number/I4LLHNSONUBE7UWDRNLXMES73F/graph.json","events_json":"https://pith.science/api/pith-number/I4LLHNSONUBE7UWDRNLXMES73F/events.json","paper":"https://pith.science/paper/I4LLHNSO"},"agent_actions":{"view_html":"https://pith.science/pith/I4LLHNSONUBE7UWDRNLXMES73F","download_json":"https://pith.science/pith/I4LLHNSONUBE7UWDRNLXMES73F.json","view_paper":"https://pith.science/paper/I4LLHNSO","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2508.10711&json=true","fetch_graph":"https://pith.science/api/pith-number/I4LLHNSONUBE7UWDRNLXMES73F/graph.json","fetch_events":"https://pith.science/api/pith-number/I4LLHNSONUBE7UWDRNLXMES73F/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/I4LLHNSONUBE7UWDRNLXMES73F/action/timestamp_anchor","attest_storage":"https://pith.science/pith/I4LLHNSONUBE7UWDRNLXMES73F/action/storage_attestation","attest_author":"https://pith.science/pith/I4LLHNSONUBE7UWDRNLXMES73F/action/author_attestation","sign_citation":"https://pith.science/pith/I4LLHNSONUBE7UWDRNLXMES73F/action/citation_signature","submit_replication":"https://pith.science/pith/I4LLHNSONUBE7UWDRNLXMES73F/action/replication_record"}},"created_at":"2026-07-05T11:55:22.299490+00:00","updated_at":"2026-07-05T11:55:22.299490+00:00"}