{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:YMK5LGUWFOZ2SXHPTBZAXRPHAK","short_pith_number":"pith:YMK5LGUW","schema_version":"1.0","canonical_sha256":"c315d59a962bb3a95cef98720bc5e702afcb4ff76063e0db2732b1c86ee4bbef","source":{"kind":"arxiv","id":"2406.07524","version":2},"attestation_state":"computed","paper":{"title":"Simple and Effective Masked Diffusion Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Aaron Gokaslan, Alexander Rush, Edgar Marroquin, Justin T Chiu, Marianne Arriola, Subham Sekhar Sahoo, Volodymyr Kuleshov, Yair Schiff","submitted_at":"2024-06-11T17:51:40Z","abstract_excerpt":"While diffusion models excel at generating high-quality images, prior work reports a significant performance gap between diffusion and autoregressive (AR) methods in language modeling. In this work, we show that simple masked discrete diffusion is more performant than previously thought. We apply an effective training recipe that improves the performance of masked diffusion models and derive a simplified, Rao-Blackwellized objective that results in additional improvements. Our objective has a simple form -- it is a mixture of classical masked language modeling losses -- and can be used to trai"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2406.07524","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-06-11T17:51:40Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"f23d13d793e1264a69fb49966a0b2f97b6c97492837e871b10da955831d4135e","abstract_canon_sha256":"1ff6a4904c916ed856edf5d8f0892ac0491d1bd80eebbcdfdb7fef4f5abb060b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:33:33.073232Z","signature_b64":"lHUFvNbRsXTJKHBTnMeushRP4OaoWnmgvJnQ8SUWyXi/Qjl3LiRIpSkGeVP23lWqY7vIFFQIUfi6Ba8ZTdyjCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"c315d59a962bb3a95cef98720bc5e702afcb4ff76063e0db2732b1c86ee4bbef","last_reissued_at":"2026-07-05T09:33:33.072758Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:33:33.072758Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Simple and Effective Masked Diffusion Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Aaron Gokaslan, Alexander Rush, Edgar Marroquin, Justin T Chiu, Marianne Arriola, Subham Sekhar Sahoo, Volodymyr Kuleshov, Yair Schiff","submitted_at":"2024-06-11T17:51:40Z","abstract_excerpt":"While diffusion models excel at generating high-quality images, prior work reports a significant performance gap between diffusion and autoregressive (AR) methods in language modeling. In this work, we show that simple masked discrete diffusion is more performant than previously thought. We apply an effective training recipe that improves the performance of masked diffusion models and derive a simplified, Rao-Blackwellized objective that results in additional improvements. Our objective has a simple form -- it is a mixture of classical masked language modeling losses -- and can be used to trai"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.07524","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.07524/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2406.07524","created_at":"2026-07-05T09:33:33.072811+00:00"},{"alias_kind":"arxiv_version","alias_value":"2406.07524v2","created_at":"2026-07-05T09:33:33.072811+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.07524","created_at":"2026-07-05T09:33:33.072811+00:00"},{"alias_kind":"pith_short_12","alias_value":"YMK5LGUWFOZ2","created_at":"2026-07-05T09:33:33.072811+00:00"},{"alias_kind":"pith_short_16","alias_value":"YMK5LGUWFOZ2SXHP","created_at":"2026-07-05T09:33:33.072811+00:00"},{"alias_kind":"pith_short_8","alias_value":"YMK5LGUW","created_at":"2026-07-05T09:33:33.072811+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":43,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.25473","citing_title":"Causal-rCM: A Unified Teacher-Forcing and Self-Forcing Open Recipe for Autoregressive Diffusion Distillation in Streaming Video Generation and Interactive World Models","ref_index":54,"is_internal_anchor":false},{"citing_arxiv_id":"2606.25331","citing_title":"Improved Large Language Diffusion Models","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2606.24119","citing_title":"When Top-1 Fails: Calibrating LoRA Monitors for Masked Diffusion LMs","ref_index":63,"is_internal_anchor":false},{"citing_arxiv_id":"2606.20299","citing_title":"Statistical Properties of Training & Generalization","ref_index":227,"is_internal_anchor":false},{"citing_arxiv_id":"2607.01436","citing_title":"Discrete Diffusion Language Models for Interactive Radiology Report Drafting","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2607.01775","citing_title":"Set Diffusion: Interpolating Token Orderings Between Autoregression and Diffusion for Fast and Flexible Decoding","ref_index":86,"is_internal_anchor":false},{"citing_arxiv_id":"2606.12807","citing_title":"Detect, Remask, Repair: Diffusion Editing for Faithful Summarization of Evolving Contexts","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2606.12841","citing_title":"TimeROME-DLM: Temporal Causal Tracing and Low-Rank Inference-Time Knowledge Editing for Masked Diffusion Language Models","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2606.08810","citing_title":"Continuous Language Diffusion as a Decoder-Interface Problem","ref_index":63,"is_internal_anchor":false},{"citing_arxiv_id":"2606.20299","citing_title":"Statistical Properties of Training & Generalization","ref_index":227,"is_internal_anchor":false},{"citing_arxiv_id":"2606.27732","citing_title":"Bifocal Diffusion Language Models: Asymmetric Bidirectional Context for Parallel Generation","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2606.30705","citing_title":"Why Do Few-Step Text Latents Fail When Image Latents Work? Non-Commitment at Sharp Categorical Readouts","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2606.29215","citing_title":"Multi-Block Diffusion Language Models","ref_index":42,"is_internal_anchor":false},{"citing_arxiv_id":"2605.22967","citing_title":"Learned Relay Representations for Forward-Thinking Discrete Diffusion Models","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2606.00091","citing_title":"DLLM-JEPA: Joint Embedding Predictive Architectures for Masked Diffusion Language Models","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2606.29215","citing_title":"Multi-Block Diffusion Language Models","ref_index":42,"is_internal_anchor":false},{"citing_arxiv_id":"2606.29150","citing_title":"Flow Reasoning Models: Scaling Reasoning Through Iterative Self-Refinement","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2605.25820","citing_title":"Visual-Redundancy-Controlled Parallel Decoding for Diffusion-Based Multimodal Large Language Models","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2606.00295","citing_title":"Adaptive Order Policies for Masked Diffusion","ref_index":83,"is_internal_anchor":false},{"citing_arxiv_id":"2605.31215","citing_title":"Fixed-Point Masked Generative Modeling","ref_index":62,"is_internal_anchor":false},{"citing_arxiv_id":"2502.17119","citing_title":"Diffusion and Flow Matching Models for Tabular Data: A Survey","ref_index":135,"is_internal_anchor":false},{"citing_arxiv_id":"2605.22967","citing_title":"Learned Relay Representations for Forward-Thinking Discrete Diffusion Models","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2605.20316","citing_title":"FullFlow: Upgrading Text-to-Image Flow Matching Models for Bidirectional Vision--Language Generation","ref_index":46,"is_internal_anchor":false},{"citing_arxiv_id":"2605.15676","citing_title":"Dynamic Chunking for Diffusion Language Models","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2605.16829","citing_title":"Constrained Code Generation with Discrete Diffusion","ref_index":22,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/YMK5LGUWFOZ2SXHPTBZAXRPHAK","json":"https://pith.science/pith/YMK5LGUWFOZ2SXHPTBZAXRPHAK.json","graph_json":"https://pith.science/api/pith-number/YMK5LGUWFOZ2SXHPTBZAXRPHAK/graph.json","events_json":"https://pith.science/api/pith-number/YMK5LGUWFOZ2SXHPTBZAXRPHAK/events.json","paper":"https://pith.science/paper/YMK5LGUW"},"agent_actions":{"view_html":"https://pith.science/pith/YMK5LGUWFOZ2SXHPTBZAXRPHAK","download_json":"https://pith.science/pith/YMK5LGUWFOZ2SXHPTBZAXRPHAK.json","view_paper":"https://pith.science/paper/YMK5LGUW","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2406.07524&json=true","fetch_graph":"https://pith.science/api/pith-number/YMK5LGUWFOZ2SXHPTBZAXRPHAK/graph.json","fetch_events":"https://pith.science/api/pith-number/YMK5LGUWFOZ2SXHPTBZAXRPHAK/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/YMK5LGUWFOZ2SXHPTBZAXRPHAK/action/timestamp_anchor","attest_storage":"https://pith.science/pith/YMK5LGUWFOZ2SXHPTBZAXRPHAK/action/storage_attestation","attest_author":"https://pith.science/pith/YMK5LGUWFOZ2SXHPTBZAXRPHAK/action/author_attestation","sign_citation":"https://pith.science/pith/YMK5LGUWFOZ2SXHPTBZAXRPHAK/action/citation_signature","submit_replication":"https://pith.science/pith/YMK5LGUWFOZ2SXHPTBZAXRPHAK/action/replication_record"}},"created_at":"2026-07-05T09:33:33.072811+00:00","updated_at":"2026-07-05T09:33:33.072811+00:00"}