{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:KG7A5XYAPKOMTMQYMKM3KHMD7Q","short_pith_number":"pith:KG7A5XYA","schema_version":"1.0","canonical_sha256":"51be0edf007a9cc9b2186299b51d83fc0d76f1ff7b274267d4250f79dc1cc4d9","source":{"kind":"arxiv","id":"2505.23660","version":1},"attestation_state":"computed","paper":{"title":"D-AR: Diffusion via Autoregressive Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Mike Zheng Shou, Ziteng Gao","submitted_at":"2025-05-29T17:09:25Z","abstract_excerpt":"This paper presents Diffusion via Autoregressive models (D-AR), a new paradigm recasting the image diffusion process as a vanilla autoregressive procedure in the standard next-token-prediction fashion. We start by designing the tokenizer that converts images into sequences of discrete tokens, where tokens in different positions can be decoded into different diffusion denoising steps in the pixel space. Thanks to the diffusion properties, these tokens naturally follow a coarse-to-fine order, which directly lends itself to autoregressive modeling. Therefore, we apply standard next-token predicti"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.23660","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2025-05-29T17:09:25Z","cross_cats_sorted":[],"title_canon_sha256":"60685e5404285752d844ea05fd381aabb935554e46b47114ec7b564ae0e08881","abstract_canon_sha256":"89f43554238375992a99cd12f466c03ceb8f65a13826638f42f3e335ad15bcef"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:12:05.699455Z","signature_b64":"72cdjEqr8vbNslFS+cFW5unYOfW61ngbkSWqoi7CGhHtdDCZn4wMLfqomKhpmqAXp2yNnkEJPfHlbqnYIquvBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"51be0edf007a9cc9b2186299b51d83fc0d76f1ff7b274267d4250f79dc1cc4d9","last_reissued_at":"2026-07-05T11:12:05.698954Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:12:05.698954Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"D-AR: Diffusion via Autoregressive Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Mike Zheng Shou, Ziteng Gao","submitted_at":"2025-05-29T17:09:25Z","abstract_excerpt":"This paper presents Diffusion via Autoregressive models (D-AR), a new paradigm recasting the image diffusion process as a vanilla autoregressive procedure in the standard next-token-prediction fashion. We start by designing the tokenizer that converts images into sequences of discrete tokens, where tokens in different positions can be decoded into different diffusion denoising steps in the pixel space. Thanks to the diffusion properties, these tokens naturally follow a coarse-to-fine order, which directly lends itself to autoregressive modeling. Therefore, we apply standard next-token predicti"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.23660","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.23660/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.23660","created_at":"2026-07-05T11:12:05.699008+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.23660v1","created_at":"2026-07-05T11:12:05.699008+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.23660","created_at":"2026-07-05T11:12:05.699008+00:00"},{"alias_kind":"pith_short_12","alias_value":"KG7A5XYAPKOM","created_at":"2026-07-05T11:12:05.699008+00:00"},{"alias_kind":"pith_short_16","alias_value":"KG7A5XYAPKOMTMQY","created_at":"2026-07-05T11:12:05.699008+00:00"},{"alias_kind":"pith_short_8","alias_value":"KG7A5XYA","created_at":"2026-07-05T11:12:05.699008+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.00329","citing_title":"Fast Text-to-Audio Generation with One-Step Sampling via Energy-Scoring and Auxiliary Contextual Representation Distillation","ref_index":9,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/KG7A5XYAPKOMTMQYMKM3KHMD7Q","json":"https://pith.science/pith/KG7A5XYAPKOMTMQYMKM3KHMD7Q.json","graph_json":"https://pith.science/api/pith-number/KG7A5XYAPKOMTMQYMKM3KHMD7Q/graph.json","events_json":"https://pith.science/api/pith-number/KG7A5XYAPKOMTMQYMKM3KHMD7Q/events.json","paper":"https://pith.science/paper/KG7A5XYA"},"agent_actions":{"view_html":"https://pith.science/pith/KG7A5XYAPKOMTMQYMKM3KHMD7Q","download_json":"https://pith.science/pith/KG7A5XYAPKOMTMQYMKM3KHMD7Q.json","view_paper":"https://pith.science/paper/KG7A5XYA","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.23660&json=true","fetch_graph":"https://pith.science/api/pith-number/KG7A5XYAPKOMTMQYMKM3KHMD7Q/graph.json","fetch_events":"https://pith.science/api/pith-number/KG7A5XYAPKOMTMQYMKM3KHMD7Q/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/KG7A5XYAPKOMTMQYMKM3KHMD7Q/action/timestamp_anchor","attest_storage":"https://pith.science/pith/KG7A5XYAPKOMTMQYMKM3KHMD7Q/action/storage_attestation","attest_author":"https://pith.science/pith/KG7A5XYAPKOMTMQYMKM3KHMD7Q/action/author_attestation","sign_citation":"https://pith.science/pith/KG7A5XYAPKOMTMQYMKM3KHMD7Q/action/citation_signature","submit_replication":"https://pith.science/pith/KG7A5XYAPKOMTMQYMKM3KHMD7Q/action/replication_record"}},"created_at":"2026-07-05T11:12:05.699008+00:00","updated_at":"2026-07-05T11:12:05.699008+00:00"}