{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:GRKZZCHM3GBTV27BLBYOFP2GFW","short_pith_number":"pith:GRKZZCHM","schema_version":"1.0","canonical_sha256":"34559c88ecd9833aebe15870e2bf462dbf122dec835bf9473b66f47221b18908","source":{"kind":"arxiv","id":"2506.09477","version":1},"attestation_state":"computed","paper":{"title":"On a few pitfalls in KL divergence gradient estimation for RL","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"R\\'emi Munos, Yunhao Tang","submitted_at":"2025-06-11T07:43:33Z","abstract_excerpt":"We point out a few pitfalls in implementing gradient estimation for KL divergence in RL training for LLM, as seen in a number of open source projects and papers. The first major pitfall is to differentiate through the KL estimate as loss functions to minimize KL divergence. We show that such implementations are generally incorrect and do not produce the desired KL gradient. Secondly, we show that some implementations do not account for the sequential nature of the estimation problem and produce a partial gradient at best. We demonstrate the impact of such issues with illustrative tabular and L"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2506.09477","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-06-11T07:43:33Z","cross_cats_sorted":[],"title_canon_sha256":"6c78758f18bf3aa728d29202a98dd54c5f828170f170827e0bb81678e6882307","abstract_canon_sha256":"6a7599b9b7d34147b9309f65bc68a792411ebb5d08cd8d345f54cf632f998754"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:19:38.098432Z","signature_b64":"GQf85IX4g7/rGkkekm7zcehdlxqUsZQLaLSrkUwmtna9pdd3z0D67wiJw52gyVR1oLOjRjfW7pAu87ibtQD+CQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"34559c88ecd9833aebe15870e2bf462dbf122dec835bf9473b66f47221b18908","last_reissued_at":"2026-07-05T11:19:38.098019Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:19:38.098019Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"On a few pitfalls in KL divergence gradient estimation for RL","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"R\\'emi Munos, Yunhao Tang","submitted_at":"2025-06-11T07:43:33Z","abstract_excerpt":"We point out a few pitfalls in implementing gradient estimation for KL divergence in RL training for LLM, as seen in a number of open source projects and papers. The first major pitfall is to differentiate through the KL estimate as loss functions to minimize KL divergence. We show that such implementations are generally incorrect and do not produce the desired KL gradient. Secondly, we show that some implementations do not account for the sequential nature of the estimation problem and produce a partial gradient at best. We demonstrate the impact of such issues with illustrative tabular and L"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.09477","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2506.09477/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2506.09477","created_at":"2026-07-05T11:19:38.098073+00:00"},{"alias_kind":"arxiv_version","alias_value":"2506.09477v1","created_at":"2026-07-05T11:19:38.098073+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.09477","created_at":"2026-07-05T11:19:38.098073+00:00"},{"alias_kind":"pith_short_12","alias_value":"GRKZZCHM3GBT","created_at":"2026-07-05T11:19:38.098073+00:00"},{"alias_kind":"pith_short_16","alias_value":"GRKZZCHM3GBTV27B","created_at":"2026-07-05T11:19:38.098073+00:00"},{"alias_kind":"pith_short_8","alias_value":"GRKZZCHM","created_at":"2026-07-05T11:19:38.098073+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":10,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.26080","citing_title":"Neglected Free Lunch from Post-training: Progress Advantage for LLM Agents","ref_index":44,"is_internal_anchor":false},{"citing_arxiv_id":"2606.03382","citing_title":"Local Guidance, Global Impact: Gaussian-Reshaped Trust Region Unlocks Behavior Transitions","ref_index":48,"is_internal_anchor":false},{"citing_arxiv_id":"2606.01039","citing_title":"OPD+: Rethinking the Advantage Design for On-Policy Distillation","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2605.21605","citing_title":"GenEvolve: Self-Evolving Image Generation Agents via Tool-Orchestrated Visual Experience Distillation","ref_index":38,"is_internal_anchor":false},{"citing_arxiv_id":"2605.21605","citing_title":"GenEvolve: Self-Evolving Image Generation Agents via Tool-Orchestrated Visual Experience Distillation","ref_index":38,"is_internal_anchor":false},{"citing_arxiv_id":"2601.16175","citing_title":"Learning to Discover at Test Time","ref_index":72,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09214","citing_title":"Fast Rates for Offline Contextual Bandits with Forward-KL Regularization under Single-Policy Concentrability","ref_index":111,"is_internal_anchor":false},{"citing_arxiv_id":"2604.10674","citing_title":"Skill-SD: Skill-Conditioned Self-Distillation for Multi-turn LLM Agents","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07865","citing_title":"KL for a KL: On-Policy Distillation with Control Variate Baseline","ref_index":39,"is_internal_anchor":false},{"citing_arxiv_id":"2604.16259","citing_title":"Beyond Distribution Sharpening: The Importance of Task Rewards","ref_index":38,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/GRKZZCHM3GBTV27BLBYOFP2GFW","json":"https://pith.science/pith/GRKZZCHM3GBTV27BLBYOFP2GFW.json","graph_json":"https://pith.science/api/pith-number/GRKZZCHM3GBTV27BLBYOFP2GFW/graph.json","events_json":"https://pith.science/api/pith-number/GRKZZCHM3GBTV27BLBYOFP2GFW/events.json","paper":"https://pith.science/paper/GRKZZCHM"},"agent_actions":{"view_html":"https://pith.science/pith/GRKZZCHM3GBTV27BLBYOFP2GFW","download_json":"https://pith.science/pith/GRKZZCHM3GBTV27BLBYOFP2GFW.json","view_paper":"https://pith.science/paper/GRKZZCHM","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2506.09477&json=true","fetch_graph":"https://pith.science/api/pith-number/GRKZZCHM3GBTV27BLBYOFP2GFW/graph.json","fetch_events":"https://pith.science/api/pith-number/GRKZZCHM3GBTV27BLBYOFP2GFW/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/GRKZZCHM3GBTV27BLBYOFP2GFW/action/timestamp_anchor","attest_storage":"https://pith.science/pith/GRKZZCHM3GBTV27BLBYOFP2GFW/action/storage_attestation","attest_author":"https://pith.science/pith/GRKZZCHM3GBTV27BLBYOFP2GFW/action/author_attestation","sign_citation":"https://pith.science/pith/GRKZZCHM3GBTV27BLBYOFP2GFW/action/citation_signature","submit_replication":"https://pith.science/pith/GRKZZCHM3GBTV27BLBYOFP2GFW/action/replication_record"}},"created_at":"2026-07-05T11:19:38.098073+00:00","updated_at":"2026-07-05T11:19:38.098073+00:00"}