{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:TGFNDEK2A3UTPHY5UAMGHBEK4M","short_pith_number":"pith:TGFNDEK2","schema_version":"1.0","canonical_sha256":"998ad1915a06e9379f1da01863848ae32f1a1807aa232fe8861deff96f96e3a9","source":{"kind":"arxiv","id":"2502.06768","version":3},"attestation_state":"computed","paper":{"title":"Train for the Worst, Plan for the Best: Understanding Token Ordering in Masked Diffusions","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Jaeyeon Kim, Kulin Shah, Sham Kakade, Sitan Chen, Vasilis Kontonis","submitted_at":"2025-02-10T18:47:21Z","abstract_excerpt":"In recent years, masked diffusion models (MDMs) have emerged as a promising alternative approach for generative modeling over discrete domains. Compared to autoregressive models (ARMs), MDMs trade off complexity at training time with flexibility at inference time. At training time, they must learn to solve an exponentially large number of infilling problems, but at inference time, they can decode tokens in essentially arbitrary order. In this work, we closely examine these two competing effects. On the training front, we theoretically and empirically demonstrate that MDMs indeed train on compu"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.06768","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-02-10T18:47:21Z","cross_cats_sorted":[],"title_canon_sha256":"7df59d63381ed1ccf87592d6f24c434ee0b5685a47d586bc0104055bab19b27e","abstract_canon_sha256":"7d8e3cda55295e3c33be635ea6466c9d3e6402606b8d36a56c3ebe2e0b7f83ca"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:56:17.508485Z","signature_b64":"o7vVobMh/JraRx5NaU9ha/XIsANd88ZztEBjV51X+yNZpO7VcgH9mMWAGZiwQzzcdOpvAUq1N4yK98eeBmdsCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"998ad1915a06e9379f1da01863848ae32f1a1807aa232fe8861deff96f96e3a9","last_reissued_at":"2026-07-05T11:56:17.507994Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:56:17.507994Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Train for the Worst, Plan for the Best: Understanding Token Ordering in Masked Diffusions","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Jaeyeon Kim, Kulin Shah, Sham Kakade, Sitan Chen, Vasilis Kontonis","submitted_at":"2025-02-10T18:47:21Z","abstract_excerpt":"In recent years, masked diffusion models (MDMs) have emerged as a promising alternative approach for generative modeling over discrete domains. Compared to autoregressive models (ARMs), MDMs trade off complexity at training time with flexibility at inference time. At training time, they must learn to solve an exponentially large number of infilling problems, but at inference time, they can decode tokens in essentially arbitrary order. In this work, we closely examine these two competing effects. On the training front, we theoretically and empirically demonstrate that MDMs indeed train on compu"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.06768","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.06768/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.06768","created_at":"2026-07-05T11:56:17.508058+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.06768v3","created_at":"2026-07-05T11:56:17.508058+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.06768","created_at":"2026-07-05T11:56:17.508058+00:00"},{"alias_kind":"pith_short_12","alias_value":"TGFNDEK2A3UT","created_at":"2026-07-05T11:56:17.508058+00:00"},{"alias_kind":"pith_short_16","alias_value":"TGFNDEK2A3UTPHY5","created_at":"2026-07-05T11:56:17.508058+00:00"},{"alias_kind":"pith_short_8","alias_value":"TGFNDEK2","created_at":"2026-07-05T11:56:17.508058+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":21,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.24773","citing_title":"Posterior Refinement: Fast Language Generation via Any-Order Flow Maps","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2606.21633","citing_title":"HERALD: High-Throughput Block Diffusion LLM Serving via CPU-GPU Cooperative KV Cache Retrieval","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2606.17999","citing_title":"VoidPadding: Let [VOID] Handle Padding in Masked Diffusion Language Models so that [EOS] Can Focus on Semantic Termination","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2607.01775","citing_title":"Set Diffusion: Interpolating Token Orderings Between Autoregression and Diffusion for Fast and Flexible Decoding","ref_index":126,"is_internal_anchor":false},{"citing_arxiv_id":"2606.15531","citing_title":"Greedy Coordinate Diffusion: Effective and Semantically Coherent Adversarial Attacks via Diffusion Guidance","ref_index":44,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07971","citing_title":"DVD: Discrete Voxel Diffusion for 3D Generation and Editing","ref_index":53,"is_internal_anchor":false},{"citing_arxiv_id":"2605.22967","citing_title":"Learned Relay Representations for Forward-Thinking Discrete Diffusion Models","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2606.29150","citing_title":"Flow Reasoning Models: Scaling Reasoning Through Iterative Self-Refinement","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2605.26106","citing_title":"Looped Diffusion Language Models","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2606.00295","citing_title":"Adaptive Order Policies for Masked Diffusion","ref_index":125,"is_internal_anchor":false},{"citing_arxiv_id":"2605.31215","citing_title":"Fixed-Point Masked Generative Modeling","ref_index":38,"is_internal_anchor":false},{"citing_arxiv_id":"2602.18176","citing_title":"Improving Sampling for Masked Diffusion Models via Information Gain","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2605.22967","citing_title":"Learned Relay Representations for Forward-Thinking Discrete Diffusion Models","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18253","citing_title":"Machine Unlearning for Masked Diffusion Language Models","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2510.03206","citing_title":"Coevolutionary Continuous Discrete Diffusion: Make Your Diffusion Language Model a Latent Reasoner","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12522","citing_title":"Differences in Text Generated by Diffusion and Autoregressive Language Models","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09397","citing_title":"BadDLM: Backdooring Diffusion Language Models with Diverse Targets","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2604.22847","citing_title":"Dream-Cubed: Controllable Generative Modeling in Minecraft by Training on Billions of Cubes","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07971","citing_title":"DVD: Discrete Voxel Diffusion for 3D Generation and Editing","ref_index":52,"is_internal_anchor":false},{"citing_arxiv_id":"2604.18471","citing_title":"NI Sampling: Accelerating Discrete Diffusion Sampling by Token Order Optimization","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2605.04291","citing_title":"Leveraging Pretrained Language Models as Energy Functions for Glauber Dynamics Text Diffusion","ref_index":5,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/TGFNDEK2A3UTPHY5UAMGHBEK4M","json":"https://pith.science/pith/TGFNDEK2A3UTPHY5UAMGHBEK4M.json","graph_json":"https://pith.science/api/pith-number/TGFNDEK2A3UTPHY5UAMGHBEK4M/graph.json","events_json":"https://pith.science/api/pith-number/TGFNDEK2A3UTPHY5UAMGHBEK4M/events.json","paper":"https://pith.science/paper/TGFNDEK2"},"agent_actions":{"view_html":"https://pith.science/pith/TGFNDEK2A3UTPHY5UAMGHBEK4M","download_json":"https://pith.science/pith/TGFNDEK2A3UTPHY5UAMGHBEK4M.json","view_paper":"https://pith.science/paper/TGFNDEK2","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.06768&json=true","fetch_graph":"https://pith.science/api/pith-number/TGFNDEK2A3UTPHY5UAMGHBEK4M/graph.json","fetch_events":"https://pith.science/api/pith-number/TGFNDEK2A3UTPHY5UAMGHBEK4M/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/TGFNDEK2A3UTPHY5UAMGHBEK4M/action/timestamp_anchor","attest_storage":"https://pith.science/pith/TGFNDEK2A3UTPHY5UAMGHBEK4M/action/storage_attestation","attest_author":"https://pith.science/pith/TGFNDEK2A3UTPHY5UAMGHBEK4M/action/author_attestation","sign_citation":"https://pith.science/pith/TGFNDEK2A3UTPHY5UAMGHBEK4M/action/citation_signature","submit_replication":"https://pith.science/pith/TGFNDEK2A3UTPHY5UAMGHBEK4M/action/replication_record"}},"created_at":"2026-07-05T11:56:17.508058+00:00","updated_at":"2026-07-05T11:56:17.508058+00:00"}