{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:JG5IS435RUCACCDSEKMA4Q4VJ7","short_pith_number":"pith:JG5IS435","schema_version":"1.0","canonical_sha256":"49ba89737d8d0401087222980e43954ff980613f313ebc7e678e6dd915c0a765","source":{"kind":"arxiv","id":"2509.10396","version":1},"attestation_state":"computed","paper":{"title":"Inpainting-Guided Policy Optimization for Diffusion Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Aditya Grover, Bo Liu, Chenyu Wang, Feiyu Chen, Guan Pang, Jing Huang, Mengchen Liu, Miao Liu, Sean Bell, Siyan Zhao, Yuandong Tian","submitted_at":"2025-09-12T16:44:31Z","abstract_excerpt":"Masked diffusion large language models (dLLMs) are emerging as promising alternatives to autoregressive LLMs, offering competitive performance while supporting unique generation capabilities such as inpainting. We explore how inpainting can inform RL algorithm design for dLLMs. Aligning LLMs with reinforcement learning faces an exploration challenge: sparse reward signals and sample waste when models fail to discover correct solutions. While this inefficiency affects LLMs broadly, dLLMs offer a distinctive opportunity--their inpainting ability can guide exploration. We introduce IGPO (Inpainti"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2509.10396","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-09-12T16:44:31Z","cross_cats_sorted":[],"title_canon_sha256":"1a24bef0dcce3215450eb9f0e5fc8a4daaab0dabe242cb85c2acef9229e304a8","abstract_canon_sha256":"a5920beb00d68dd1e509b046be7c76fe3b5122dafd5b3cd73cedb66f4be68216"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T12:11:09.131494Z","signature_b64":"6W7KG3W8HX+mVrhCtGPCNF54840C8C5dxOdVagMP78dCVEsW2XMK4rk7vtrYKr2Sks0Bc7ERXP6Fe7apLPytCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"49ba89737d8d0401087222980e43954ff980613f313ebc7e678e6dd915c0a765","last_reissued_at":"2026-07-05T12:11:09.130947Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T12:11:09.130947Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Inpainting-Guided Policy Optimization for Diffusion Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Aditya Grover, Bo Liu, Chenyu Wang, Feiyu Chen, Guan Pang, Jing Huang, Mengchen Liu, Miao Liu, Sean Bell, Siyan Zhao, Yuandong Tian","submitted_at":"2025-09-12T16:44:31Z","abstract_excerpt":"Masked diffusion large language models (dLLMs) are emerging as promising alternatives to autoregressive LLMs, offering competitive performance while supporting unique generation capabilities such as inpainting. We explore how inpainting can inform RL algorithm design for dLLMs. Aligning LLMs with reinforcement learning faces an exploration challenge: sparse reward signals and sample waste when models fail to discover correct solutions. While this inefficiency affects LLMs broadly, dLLMs offer a distinctive opportunity--their inpainting ability can guide exploration. We introduce IGPO (Inpainti"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2509.10396","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2509.10396/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2509.10396","created_at":"2026-07-05T12:11:09.131007+00:00"},{"alias_kind":"arxiv_version","alias_value":"2509.10396v1","created_at":"2026-07-05T12:11:09.131007+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2509.10396","created_at":"2026-07-05T12:11:09.131007+00:00"},{"alias_kind":"pith_short_12","alias_value":"JG5IS435RUCA","created_at":"2026-07-05T12:11:09.131007+00:00"},{"alias_kind":"pith_short_16","alias_value":"JG5IS435RUCACCDS","created_at":"2026-07-05T12:11:09.131007+00:00"},{"alias_kind":"pith_short_8","alias_value":"JG5IS435","created_at":"2026-07-05T12:11:09.131007+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.17497","citing_title":"Self-Supervised On-Policy Distillation for Reasoning Language Models","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2601.18734","citing_title":"Self-Distilled Reasoner: On-Policy Self-Distillation for Large Language Models","ref_index":25,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/JG5IS435RUCACCDSEKMA4Q4VJ7","json":"https://pith.science/pith/JG5IS435RUCACCDSEKMA4Q4VJ7.json","graph_json":"https://pith.science/api/pith-number/JG5IS435RUCACCDSEKMA4Q4VJ7/graph.json","events_json":"https://pith.science/api/pith-number/JG5IS435RUCACCDSEKMA4Q4VJ7/events.json","paper":"https://pith.science/paper/JG5IS435"},"agent_actions":{"view_html":"https://pith.science/pith/JG5IS435RUCACCDSEKMA4Q4VJ7","download_json":"https://pith.science/pith/JG5IS435RUCACCDSEKMA4Q4VJ7.json","view_paper":"https://pith.science/paper/JG5IS435","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2509.10396&json=true","fetch_graph":"https://pith.science/api/pith-number/JG5IS435RUCACCDSEKMA4Q4VJ7/graph.json","fetch_events":"https://pith.science/api/pith-number/JG5IS435RUCACCDSEKMA4Q4VJ7/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/JG5IS435RUCACCDSEKMA4Q4VJ7/action/timestamp_anchor","attest_storage":"https://pith.science/pith/JG5IS435RUCACCDSEKMA4Q4VJ7/action/storage_attestation","attest_author":"https://pith.science/pith/JG5IS435RUCACCDSEKMA4Q4VJ7/action/author_attestation","sign_citation":"https://pith.science/pith/JG5IS435RUCACCDSEKMA4Q4VJ7/action/citation_signature","submit_replication":"https://pith.science/pith/JG5IS435RUCACCDSEKMA4Q4VJ7/action/replication_record"}},"created_at":"2026-07-05T12:11:09.131007+00:00","updated_at":"2026-07-05T12:11:09.131007+00:00"}