{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:ULCPVUMBZA3IKOFD5V3ACW6LKT","short_pith_number":"pith:ULCPVUMB","schema_version":"1.0","canonical_sha256":"a2c4fad181c8368538a3ed76015bcb54d05a9ca186a8a45fd67cb43ae69541da","source":{"kind":"arxiv","id":"2504.12216","version":2},"attestation_state":"computed","paper":{"title":"d1: Scaling Reasoning in Diffusion Large Language Models via Reinforcement Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Aditya Grover, Devaansh Gupta, Qinqing Zheng, Siyan Zhao","submitted_at":"2025-04-16T16:08:45Z","abstract_excerpt":"Recent large language models (LLMs) have demonstrated strong reasoning capabilities that benefits from online reinforcement learning (RL). These capabilities have primarily been demonstrated within the left-to-right autoregressive (AR) generation paradigm. In contrast, non-autoregressive paradigms based on diffusion generate text in a coarse-to-fine manner. Although recent diffusion-based large language models (dLLMs) have achieved competitive language modeling performance compared to their AR counterparts, it remains unclear if dLLMs can also leverage recent advances in LLM reasoning. To this"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2504.12216","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-04-16T16:08:45Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"908a2839fc18992a716a5893d258f9a72903d87544703c2a1240feed8709899c","abstract_canon_sha256":"fb36e349cd07f17d75fa08f89e6df3a38b7694a6cb3a8bb4da125125d3175fa4"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:14:58.308721Z","signature_b64":"wP9PQPl1NuDRvJBXdJjnY8/8dlbOqgDniuA14H3rijune5jZGF1fAOSTYELVNEgYeRSI3GVJ2eDUPLUuHqfACA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"a2c4fad181c8368538a3ed76015bcb54d05a9ca186a8a45fd67cb43ae69541da","last_reissued_at":"2026-07-05T11:14:58.308214Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:14:58.308214Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"d1: Scaling Reasoning in Diffusion Large Language Models via Reinforcement Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Aditya Grover, Devaansh Gupta, Qinqing Zheng, Siyan Zhao","submitted_at":"2025-04-16T16:08:45Z","abstract_excerpt":"Recent large language models (LLMs) have demonstrated strong reasoning capabilities that benefits from online reinforcement learning (RL). These capabilities have primarily been demonstrated within the left-to-right autoregressive (AR) generation paradigm. In contrast, non-autoregressive paradigms based on diffusion generate text in a coarse-to-fine manner. Although recent diffusion-based large language models (dLLMs) have achieved competitive language modeling performance compared to their AR counterparts, it remains unclear if dLLMs can also leverage recent advances in LLM reasoning. To this"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2504.12216","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2504.12216/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2504.12216","created_at":"2026-07-05T11:14:58.308270+00:00"},{"alias_kind":"arxiv_version","alias_value":"2504.12216v2","created_at":"2026-07-05T11:14:58.308270+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2504.12216","created_at":"2026-07-05T11:14:58.308270+00:00"},{"alias_kind":"pith_short_12","alias_value":"ULCPVUMBZA3I","created_at":"2026-07-05T11:14:58.308270+00:00"},{"alias_kind":"pith_short_16","alias_value":"ULCPVUMBZA3IKOFD","created_at":"2026-07-05T11:14:58.308270+00:00"},{"alias_kind":"pith_short_8","alias_value":"ULCPVUMB","created_at":"2026-07-05T11:14:58.308270+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":38,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.05722","citing_title":"Nemotron-Labs-Diffusion: A Tri-Mode Language Model Unifying Autoregressive, Diffusion, and Self-Speculation Decoding","ref_index":57,"is_internal_anchor":true},{"citing_arxiv_id":"2606.25331","citing_title":"Improved Large Language Diffusion Models","ref_index":46,"is_internal_anchor":false},{"citing_arxiv_id":"2606.18394","citing_title":"JetSpec: Breaking the Scaling Ceiling of Speculative Decoding with Parallel Tree Drafting","ref_index":44,"is_internal_anchor":false},{"citing_arxiv_id":"2606.18195","citing_title":"Learning from the Self-future: On-policy Self-distillation for dLLMs","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2606.13795","citing_title":"DiPOD: Diffusion Policy Optimization without Drifting Apart","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2606.12273","citing_title":"Beyond Fully Random Masking: Attention-Guided Denoising and Optimization for Diffusion Language Models","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2606.12232","citing_title":"Re-evaluating Confidence Remasking in Masked Diffusion Language Models","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2606.08501","citing_title":"Back on Track: Aligning Rewards and States for Reasoning in Diffusion Large Language Models","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2606.04535","citing_title":"Dynamic Infilling Anchors for Format-Constrained Generation in Diffusion Large Language Models","ref_index":83,"is_internal_anchor":false},{"citing_arxiv_id":"2606.04027","citing_title":"MaskForge: Structure-Aware Adaptive Attacks for Jailbreaking Diffusion Large Language Models","ref_index":52,"is_internal_anchor":false},{"citing_arxiv_id":"2605.02263","citing_title":"Break the Block: Dynamic-size Reasoning Blocks for Diffusion Large Language Models via Monotonic Entropy Descent with Reinforcement Learning","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2605.25638","citing_title":"Reinforcement Learning from Denoising Feedback","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2605.29398","citing_title":"GDSD: Reinforcement Learning as Guided Denoiser Self-Distillation for Diffusion Language Models","ref_index":62,"is_internal_anchor":false},{"citing_arxiv_id":"2605.30876","citing_title":"dMoE: dLLMs with Learnable Block Experts","ref_index":52,"is_internal_anchor":false},{"citing_arxiv_id":"2605.30753","citing_title":"Efficient Diffusion LLMs via Temporal-Spatial Parallel Decoding and Confidence Extrapolation","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2605.23346","citing_title":"Contrastive Distribution Matching for Amortized Sequential Monte Carlo in Discrete Diffusion","ref_index":94,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11854","citing_title":"Self-Distilled Trajectory-Aware Boltzmann Modeling: Bridging the Training-Inference Discrepancy in Diffusion Language Models","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17174","citing_title":"Beyond Execution: Static-Analysis Rewards and Hint-Conditioned Diffusion RL for Code Generation","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18165","citing_title":"Elastic-dLLM: Position Preserving Context Compression and Augmentation of Diffusion LLMs","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2605.16842","citing_title":"Sketch Then Paint: Hierarchical Reinforcement Learning for Diffusion Multi-Modal Large Language Models","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2605.16941","citing_title":"Roll Out and Roll Back: Diffusion LLMs are Their Own Efficiency Teachers","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2604.08302","citing_title":"DMax: Aggressive Parallel Decoding for dLLMs","ref_index":103,"is_internal_anchor":false},{"citing_arxiv_id":"2509.20863","citing_title":"GIFT: Guided Importance-Aware Fine-Tuning for Diffusion Language Models","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2509.21912","citing_title":"Discrete Guidance Matching: Exact Guidance for Discrete Flow Matching","ref_index":87,"is_internal_anchor":false},{"citing_arxiv_id":"2512.14067","citing_title":"Efficient-DLM: From Autoregressive to Diffusion Language Models, and Beyond in Speed","ref_index":35,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ULCPVUMBZA3IKOFD5V3ACW6LKT","json":"https://pith.science/pith/ULCPVUMBZA3IKOFD5V3ACW6LKT.json","graph_json":"https://pith.science/api/pith-number/ULCPVUMBZA3IKOFD5V3ACW6LKT/graph.json","events_json":"https://pith.science/api/pith-number/ULCPVUMBZA3IKOFD5V3ACW6LKT/events.json","paper":"https://pith.science/paper/ULCPVUMB"},"agent_actions":{"view_html":"https://pith.science/pith/ULCPVUMBZA3IKOFD5V3ACW6LKT","download_json":"https://pith.science/pith/ULCPVUMBZA3IKOFD5V3ACW6LKT.json","view_paper":"https://pith.science/paper/ULCPVUMB","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2504.12216&json=true","fetch_graph":"https://pith.science/api/pith-number/ULCPVUMBZA3IKOFD5V3ACW6LKT/graph.json","fetch_events":"https://pith.science/api/pith-number/ULCPVUMBZA3IKOFD5V3ACW6LKT/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ULCPVUMBZA3IKOFD5V3ACW6LKT/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ULCPVUMBZA3IKOFD5V3ACW6LKT/action/storage_attestation","attest_author":"https://pith.science/pith/ULCPVUMBZA3IKOFD5V3ACW6LKT/action/author_attestation","sign_citation":"https://pith.science/pith/ULCPVUMBZA3IKOFD5V3ACW6LKT/action/citation_signature","submit_replication":"https://pith.science/pith/ULCPVUMBZA3IKOFD5V3ACW6LKT/action/replication_record"}},"created_at":"2026-07-05T11:14:58.308270+00:00","updated_at":"2026-07-05T11:14:58.308270+00:00"}