{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:HBOV2F7K3BTY5X7HJLLDREZDWW","short_pith_number":"pith:HBOV2F7K","schema_version":"1.0","canonical_sha256":"385d5d17ead8678edfe74ad6389323b5aa418340f00b0c931ef13bc3fb0a3028","source":{"kind":"arxiv","id":"2406.04329","version":4},"attestation_state":"computed","paper":{"title":"Simplified and Generalized Masked Diffusion for Discrete Data","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["stat.ML"],"primary_cat":"cs.LG","authors_text":"Arnaud Doucet, Jiaxin Shi, Kehang Han, Michalis K. Titsias, Zhe Wang","submitted_at":"2024-06-06T17:59:10Z","abstract_excerpt":"Masked (or absorbing) diffusion is actively explored as an alternative to autoregressive models for generative modeling of discrete data. However, existing work in this area has been hindered by unnecessarily complex model formulations and unclear relationships between different perspectives, leading to suboptimal parameterization, training objectives, and ad hoc adjustments to counteract these issues. In this work, we aim to provide a simple and general framework that unlocks the full potential of masked diffusion models. We show that the continuous-time variational objective of masked diffus"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2406.04329","kind":"arxiv","version":4},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-06-06T17:59:10Z","cross_cats_sorted":["stat.ML"],"title_canon_sha256":"c84b05b59c22c7f599de71f9dfd1c1df5c2969fe069c3d003cb906b030715b37","abstract_canon_sha256":"35ebab9af072266cf3818d9fe49200a653ab375f7206f9a8735f40294e3d332c"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:01:33.863111Z","signature_b64":"qCM3F4/r5BGDTRVppAV3+IkBMRchdTrrz2scNgFFbs28EoALpoZt3GRFupYGd1rAzZFUnmmqKwlkuFOk3XiuDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"385d5d17ead8678edfe74ad6389323b5aa418340f00b0c931ef13bc3fb0a3028","last_reissued_at":"2026-07-05T10:01:33.862326Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:01:33.862326Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Simplified and Generalized Masked Diffusion for Discrete Data","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["stat.ML"],"primary_cat":"cs.LG","authors_text":"Arnaud Doucet, Jiaxin Shi, Kehang Han, Michalis K. Titsias, Zhe Wang","submitted_at":"2024-06-06T17:59:10Z","abstract_excerpt":"Masked (or absorbing) diffusion is actively explored as an alternative to autoregressive models for generative modeling of discrete data. However, existing work in this area has been hindered by unnecessarily complex model formulations and unclear relationships between different perspectives, leading to suboptimal parameterization, training objectives, and ad hoc adjustments to counteract these issues. In this work, we aim to provide a simple and general framework that unlocks the full potential of masked diffusion models. We show that the continuous-time variational objective of masked diffus"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.04329","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.04329/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2406.04329","created_at":"2026-07-05T10:01:33.862652+00:00"},{"alias_kind":"arxiv_version","alias_value":"2406.04329v4","created_at":"2026-07-05T10:01:33.862652+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.04329","created_at":"2026-07-05T10:01:33.862652+00:00"},{"alias_kind":"pith_short_12","alias_value":"HBOV2F7K3BTY","created_at":"2026-07-05T10:01:33.862652+00:00"},{"alias_kind":"pith_short_16","alias_value":"HBOV2F7K3BTY5X7H","created_at":"2026-07-05T10:01:33.862652+00:00"},{"alias_kind":"pith_short_8","alias_value":"HBOV2F7K","created_at":"2026-07-05T10:01:33.862652+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":26,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.25473","citing_title":"Causal-rCM: A Unified Teacher-Forcing and Self-Forcing Open Recipe for Autoregressive Diffusion Distillation in Streaming Video Generation and Interactive World Models","ref_index":57,"is_internal_anchor":false},{"citing_arxiv_id":"2606.25331","citing_title":"Improved Large Language Diffusion Models","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2606.27361","citing_title":"Autoregressive Boltzmann Generators","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2606.08417","citing_title":"Hacking Generative Perplexity: Why Unconditional Text Evaluation Needs Distributional Metrics","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2606.27732","citing_title":"Bifocal Diffusion Language Models: Asymmetric Bidirectional Context for Parallel Generation","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2605.22967","citing_title":"Learned Relay Representations for Forward-Thinking Discrete Diffusion Models","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2605.29398","citing_title":"GDSD: Reinforcement Learning as Guided Denoiser Self-Distillation for Diffusion Language Models","ref_index":46,"is_internal_anchor":false},{"citing_arxiv_id":"2606.00295","citing_title":"Adaptive Order Policies for Masked Diffusion","ref_index":84,"is_internal_anchor":false},{"citing_arxiv_id":"2605.31215","citing_title":"Fixed-Point Masked Generative Modeling","ref_index":67,"is_internal_anchor":false},{"citing_arxiv_id":"2606.26566","citing_title":"Adversarial Diffusion Across Modalities: A Fusion Survey of Attacks, Defenses, and Evaluation for Text, Vision, and Vision-Language Models","ref_index":60,"is_internal_anchor":false},{"citing_arxiv_id":"2502.17119","citing_title":"Diffusion and Flow Matching Models for Tabular Data: A Survey","ref_index":143,"is_internal_anchor":false},{"citing_arxiv_id":"2605.22967","citing_title":"Learned Relay Representations for Forward-Thinking Discrete Diffusion Models","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11854","citing_title":"Self-Distilled Trajectory-Aware Boltzmann Modeling: Bridging the Training-Inference Discrepancy in Diffusion Language Models","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2410.17891","citing_title":"Scaling Diffusion Language Models via Adaptation from Autoregressive Models","ref_index":177,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18204","citing_title":"Forward-Learned Discrete Diffusion: Learning how to noise to denoise faster","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2508.19982","citing_title":"Diffusion Language Models Know the Answer Before Decoding","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2505.16933","citing_title":"LLaDA-V: Large Language Diffusion Models with Visual Instruction Tuning","ref_index":39,"is_internal_anchor":false},{"citing_arxiv_id":"2604.02340","citing_title":"Not All Denoising Steps Are Equal: Model Scheduling for Faster Masked Diffusion Language Models","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2505.22618","citing_title":"Fast-dLLM: Training-free Acceleration of Diffusion LLM by Enabling KV Cache and Parallel Decoding","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2508.02193","citing_title":"Seed Diffusion: A Large-Scale Diffusion Language Model with High-Speed Inference","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11726","citing_title":"Block-R1: Rethinking the Role of Block Size in Multi-domain Reinforcement Learning for Diffusion Large Language Models","ref_index":50,"is_internal_anchor":false},{"citing_arxiv_id":"2604.02718","citing_title":"Generative Frontiers: Why Evaluation Matters for Diffusion Language Models","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11726","citing_title":"Block-R1: Rethinking the Role of Block Size in Multi-domain Reinforcement Learning for Diffusion Large Language Models","ref_index":50,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11854","citing_title":"Self-Distilled Trajectory-Aware Boltzmann Modeling: Bridging the Training-Inference Discrepancy in Diffusion Language Models","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2605.00354","citing_title":"VQ-SAD: Vector Quantized Structure Aware Diffusion For Molecule Generation","ref_index":8,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/HBOV2F7K3BTY5X7HJLLDREZDWW","json":"https://pith.science/pith/HBOV2F7K3BTY5X7HJLLDREZDWW.json","graph_json":"https://pith.science/api/pith-number/HBOV2F7K3BTY5X7HJLLDREZDWW/graph.json","events_json":"https://pith.science/api/pith-number/HBOV2F7K3BTY5X7HJLLDREZDWW/events.json","paper":"https://pith.science/paper/HBOV2F7K"},"agent_actions":{"view_html":"https://pith.science/pith/HBOV2F7K3BTY5X7HJLLDREZDWW","download_json":"https://pith.science/pith/HBOV2F7K3BTY5X7HJLLDREZDWW.json","view_paper":"https://pith.science/paper/HBOV2F7K","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2406.04329&json=true","fetch_graph":"https://pith.science/api/pith-number/HBOV2F7K3BTY5X7HJLLDREZDWW/graph.json","fetch_events":"https://pith.science/api/pith-number/HBOV2F7K3BTY5X7HJLLDREZDWW/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/HBOV2F7K3BTY5X7HJLLDREZDWW/action/timestamp_anchor","attest_storage":"https://pith.science/pith/HBOV2F7K3BTY5X7HJLLDREZDWW/action/storage_attestation","attest_author":"https://pith.science/pith/HBOV2F7K3BTY5X7HJLLDREZDWW/action/author_attestation","sign_citation":"https://pith.science/pith/HBOV2F7K3BTY5X7HJLLDREZDWW/action/citation_signature","submit_replication":"https://pith.science/pith/HBOV2F7K3BTY5X7HJLLDREZDWW/action/replication_record"}},"created_at":"2026-07-05T10:01:33.862652+00:00","updated_at":"2026-07-05T10:01:33.862652+00:00"}