{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:JHF2OAQSHQQHZZBILL36ZEZIYN","short_pith_number":"pith:JHF2OAQS","schema_version":"1.0","canonical_sha256":"49cba702123c207ce4285af7ec9328c361578721b171a8f9b46be8cf5968a919","source":{"kind":"arxiv","id":"2409.02908","version":6},"attestation_state":"computed","paper":{"title":"Masked Diffusion Models are Secretly Time-Agnostic Masked Models and Exploit Inaccurate Categorical Sampling","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Hanzi Mao, Jun Zhu, Kaiwen Zheng, Ming-Yu Liu, Qinsheng Zhang, Yongxin Chen","submitted_at":"2024-09-04T17:48:19Z","abstract_excerpt":"Masked diffusion models (MDMs) have emerged as a popular research topic for generative modeling of discrete data, thanks to their superior performance over other discrete diffusion models, and are rivaling the auto-regressive models (ARMs) for language modeling tasks. The recent effort in simplifying the masked diffusion framework further leads to alignment with continuous-space diffusion models and more principled training and sampling recipes. In this paper, however, we reveal that both training and sampling of MDMs are theoretically free from the time variable, arguably the key signature of"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2409.02908","kind":"arxiv","version":6},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-09-04T17:48:19Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"bf56c184ead1d02cce00bd4179a870831683e74c9000a8db48e6719e88a8daa8","abstract_canon_sha256":"cde1de275b93798076204e03c1bb3943c9c40fd4cbeec2824aa8fc33bbb80c38"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:56:08.685010Z","signature_b64":"lcX607zdHwsTVRFLBG162cksBmMjcFMM39xKHaFxU7LRUBtInvCVQ7cgO35XRSjtyWBruOWMX6wwvQxnVcI/AA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"49cba702123c207ce4285af7ec9328c361578721b171a8f9b46be8cf5968a919","last_reissued_at":"2026-07-05T10:56:08.684485Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:56:08.684485Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Masked Diffusion Models are Secretly Time-Agnostic Masked Models and Exploit Inaccurate Categorical Sampling","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Hanzi Mao, Jun Zhu, Kaiwen Zheng, Ming-Yu Liu, Qinsheng Zhang, Yongxin Chen","submitted_at":"2024-09-04T17:48:19Z","abstract_excerpt":"Masked diffusion models (MDMs) have emerged as a popular research topic for generative modeling of discrete data, thanks to their superior performance over other discrete diffusion models, and are rivaling the auto-regressive models (ARMs) for language modeling tasks. The recent effort in simplifying the masked diffusion framework further leads to alignment with continuous-space diffusion models and more principled training and sampling recipes. In this paper, however, we reveal that both training and sampling of MDMs are theoretically free from the time variable, arguably the key signature of"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2409.02908","kind":"arxiv","version":6},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2409.02908/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2409.02908","created_at":"2026-07-05T10:56:08.684553+00:00"},{"alias_kind":"arxiv_version","alias_value":"2409.02908v6","created_at":"2026-07-05T10:56:08.684553+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2409.02908","created_at":"2026-07-05T10:56:08.684553+00:00"},{"alias_kind":"pith_short_12","alias_value":"JHF2OAQSHQQH","created_at":"2026-07-05T10:56:08.684553+00:00"},{"alias_kind":"pith_short_16","alias_value":"JHF2OAQSHQQHZZBI","created_at":"2026-07-05T10:56:08.684553+00:00"},{"alias_kind":"pith_short_8","alias_value":"JHF2OAQS","created_at":"2026-07-05T10:56:08.684553+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":27,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.24773","citing_title":"Posterior Refinement: Fast Language Generation via Any-Order Flow Maps","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2607.01775","citing_title":"Set Diffusion: Interpolating Token Orderings Between Autoregression and Diffusion for Fast and Flexible Decoding","ref_index":95,"is_internal_anchor":false},{"citing_arxiv_id":"2607.00773","citing_title":"Accelerating Discrete Diffusion Models with Parallel-In-Time Sampling","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2606.27617","citing_title":"Masked Language Flow Models","ref_index":47,"is_internal_anchor":false},{"citing_arxiv_id":"2605.22967","citing_title":"Learned Relay Representations for Forward-Thinking Discrete Diffusion Models","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2605.31215","citing_title":"Fixed-Point Masked Generative Modeling","ref_index":76,"is_internal_anchor":false},{"citing_arxiv_id":"2605.30876","citing_title":"dMoE: dLLMs with Learnable Block Experts","ref_index":50,"is_internal_anchor":false},{"citing_arxiv_id":"2606.06031","citing_title":"NAVIRA: Decoupled Stochastic Remasking for Masked Diffusion Language Models","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2606.08810","citing_title":"Continuous Language Diffusion as a Decoder-Interface Problem","ref_index":87,"is_internal_anchor":false},{"citing_arxiv_id":"2605.22967","citing_title":"Learned Relay Representations for Forward-Thinking Discrete Diffusion Models","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2605.22765","citing_title":"Uniform Diffusion Models Revisited: Leave-One-Out Denoiser and Absorbing State Reformulation","ref_index":41,"is_internal_anchor":false},{"citing_arxiv_id":"2602.16813","citing_title":"Flow Map Language Models: One-step Language Modeling via Continuous Denoising","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11854","citing_title":"Self-Distilled Trajectory-Aware Boltzmann Modeling: Bridging the Training-Inference Discrepancy in Diffusion Language Models","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2604.08302","citing_title":"DMax: Aggressive Parallel Decoding for dLLMs","ref_index":105,"is_internal_anchor":false},{"citing_arxiv_id":"2508.19982","citing_title":"Diffusion Language Models Know the Answer Before Decoding","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2505.16933","citing_title":"LLaDA-V: Large Language Diffusion Models with Visual Instruction Tuning","ref_index":110,"is_internal_anchor":false},{"citing_arxiv_id":"2505.22618","citing_title":"Fast-dLLM: Training-free Acceleration of Diffusion LLM by Enabling KV Cache and Parallel Decoding","ref_index":41,"is_internal_anchor":false},{"citing_arxiv_id":"2602.16813","citing_title":"Flow Map Language Models: One-step Language Modeling via Continuous Denoising","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2603.22241","citing_title":"MemDLM: Memory-Enhanced DLM Training","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2604.02718","citing_title":"Generative Frontiers: Why Evaluation Matters for Diffusion Language Models","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11854","citing_title":"Self-Distilled Trajectory-Aware Boltzmann Modeling: Bridging the Training-Inference Discrepancy in Diffusion Language Models","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2604.26985","citing_title":"Simple Self-Conditioning Adaptation for Masked Diffusion Models","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06548","citing_title":"Continuous Latent Diffusion Language Model","ref_index":114,"is_internal_anchor":false},{"citing_arxiv_id":"2604.18739","citing_title":"Discrete Tilt Matching","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2604.08302","citing_title":"DMax: Aggressive Parallel Decoding for dLLMs","ref_index":105,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/JHF2OAQSHQQHZZBILL36ZEZIYN","json":"https://pith.science/pith/JHF2OAQSHQQHZZBILL36ZEZIYN.json","graph_json":"https://pith.science/api/pith-number/JHF2OAQSHQQHZZBILL36ZEZIYN/graph.json","events_json":"https://pith.science/api/pith-number/JHF2OAQSHQQHZZBILL36ZEZIYN/events.json","paper":"https://pith.science/paper/JHF2OAQS"},"agent_actions":{"view_html":"https://pith.science/pith/JHF2OAQSHQQHZZBILL36ZEZIYN","download_json":"https://pith.science/pith/JHF2OAQSHQQHZZBILL36ZEZIYN.json","view_paper":"https://pith.science/paper/JHF2OAQS","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2409.02908&json=true","fetch_graph":"https://pith.science/api/pith-number/JHF2OAQSHQQHZZBILL36ZEZIYN/graph.json","fetch_events":"https://pith.science/api/pith-number/JHF2OAQSHQQHZZBILL36ZEZIYN/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/JHF2OAQSHQQHZZBILL36ZEZIYN/action/timestamp_anchor","attest_storage":"https://pith.science/pith/JHF2OAQSHQQHZZBILL36ZEZIYN/action/storage_attestation","attest_author":"https://pith.science/pith/JHF2OAQSHQQHZZBILL36ZEZIYN/action/author_attestation","sign_citation":"https://pith.science/pith/JHF2OAQSHQQHZZBILL36ZEZIYN/action/citation_signature","submit_replication":"https://pith.science/pith/JHF2OAQSHQQHZZBILL36ZEZIYN/action/replication_record"}},"created_at":"2026-07-05T10:56:08.684553+00:00","updated_at":"2026-07-05T10:56:08.684553+00:00"}