{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:4KJPN6ET3OUGDF6PVTXWCVE2KL","short_pith_number":"pith:4KJPN6ET","schema_version":"1.0","canonical_sha256":"e292f6f893dba86197cfacef61549a52eada64092263304b7b11987c8c9137cb","source":{"kind":"arxiv","id":"2502.09622","version":2},"attestation_state":"computed","paper":{"title":"Theoretical Benefit and Limitation of Diffusion Language Model","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL","stat.ML"],"primary_cat":"cs.LG","authors_text":"Di He, Guhao Feng, Jian Guan, Liwei Wang, Wei Wu, Yihan Geng","submitted_at":"2025-02-13T18:59:47Z","abstract_excerpt":"Diffusion language models have emerged as a promising approach for text generation. One would naturally expect this method to be an efficient replacement for autoregressive models since multiple tokens can be sampled in parallel during each diffusion step. However, its efficiency-accuracy trade-off is not yet well understood. In this paper, we present a rigorous theoretical analysis of a widely used type of diffusion language model, the Masked Diffusion Model (MDM), and find that its effectiveness heavily depends on the target evaluation metric. Under mild conditions, we prove that when using "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.09622","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2025-02-13T18:59:47Z","cross_cats_sorted":["cs.AI","cs.CL","stat.ML"],"title_canon_sha256":"41f5082257f5696d90fd4c541e6e8703612f3f2eba6d21b130347d5453f58e35","abstract_canon_sha256":"98103fc7594f78fd2346081e1221bc3a89324fb8440e1d120debd8159e001fbe"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:18:17.999839Z","signature_b64":"dNsfie/UoKknXMTDSQC4Yn8HS2PKJrC9LmMuXrW6BCYLDJxvmggvv3gn0oEFOENo/jXrxTIX97kzELaw0ALVAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"e292f6f893dba86197cfacef61549a52eada64092263304b7b11987c8c9137cb","last_reissued_at":"2026-07-05T11:18:17.999360Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:18:17.999360Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Theoretical Benefit and Limitation of Diffusion Language Model","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL","stat.ML"],"primary_cat":"cs.LG","authors_text":"Di He, Guhao Feng, Jian Guan, Liwei Wang, Wei Wu, Yihan Geng","submitted_at":"2025-02-13T18:59:47Z","abstract_excerpt":"Diffusion language models have emerged as a promising approach for text generation. One would naturally expect this method to be an efficient replacement for autoregressive models since multiple tokens can be sampled in parallel during each diffusion step. However, its efficiency-accuracy trade-off is not yet well understood. In this paper, we present a rigorous theoretical analysis of a widely used type of diffusion language model, the Masked Diffusion Model (MDM), and find that its effectiveness heavily depends on the target evaluation metric. Under mild conditions, we prove that when using "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.09622","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.09622/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.09622","created_at":"2026-07-05T11:18:17.999418+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.09622v2","created_at":"2026-07-05T11:18:17.999418+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.09622","created_at":"2026-07-05T11:18:17.999418+00:00"},{"alias_kind":"pith_short_12","alias_value":"4KJPN6ET3OUG","created_at":"2026-07-05T11:18:17.999418+00:00"},{"alias_kind":"pith_short_16","alias_value":"4KJPN6ET3OUGDF6P","created_at":"2026-07-05T11:18:17.999418+00:00"},{"alias_kind":"pith_short_8","alias_value":"4KJPN6ET","created_at":"2026-07-05T11:18:17.999418+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2510.06133","citing_title":"CreditDecoding: Accelerating Parallel Decoding in Diffusion Large Language Models with Trace Credit","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12522","citing_title":"Differences in Text Generated by Diffusion and Autoregressive Language Models","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2604.15009","citing_title":"Towards Faster Language Model Inference Using Mixture-of-Experts Flow Matching","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2604.17068","citing_title":"Stability-Weighted Decoding for Diffusion Language Models","ref_index":5,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/4KJPN6ET3OUGDF6PVTXWCVE2KL","json":"https://pith.science/pith/4KJPN6ET3OUGDF6PVTXWCVE2KL.json","graph_json":"https://pith.science/api/pith-number/4KJPN6ET3OUGDF6PVTXWCVE2KL/graph.json","events_json":"https://pith.science/api/pith-number/4KJPN6ET3OUGDF6PVTXWCVE2KL/events.json","paper":"https://pith.science/paper/4KJPN6ET"},"agent_actions":{"view_html":"https://pith.science/pith/4KJPN6ET3OUGDF6PVTXWCVE2KL","download_json":"https://pith.science/pith/4KJPN6ET3OUGDF6PVTXWCVE2KL.json","view_paper":"https://pith.science/paper/4KJPN6ET","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.09622&json=true","fetch_graph":"https://pith.science/api/pith-number/4KJPN6ET3OUGDF6PVTXWCVE2KL/graph.json","fetch_events":"https://pith.science/api/pith-number/4KJPN6ET3OUGDF6PVTXWCVE2KL/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/4KJPN6ET3OUGDF6PVTXWCVE2KL/action/timestamp_anchor","attest_storage":"https://pith.science/pith/4KJPN6ET3OUGDF6PVTXWCVE2KL/action/storage_attestation","attest_author":"https://pith.science/pith/4KJPN6ET3OUGDF6PVTXWCVE2KL/action/author_attestation","sign_citation":"https://pith.science/pith/4KJPN6ET3OUGDF6PVTXWCVE2KL/action/citation_signature","submit_replication":"https://pith.science/pith/4KJPN6ET3OUGDF6PVTXWCVE2KL/action/replication_record"}},"created_at":"2026-07-05T11:18:17.999418+00:00","updated_at":"2026-07-05T11:18:17.999418+00:00"}