{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2026:PB6EH4S2MCUDJM7PRCWXUXC2ZB","short_pith_number":"pith:PB6EH4S2","schema_version":"1.0","canonical_sha256":"787c43f25a60a834b3ef88ad7a5c5ac84f88c4794789fa85bb8028e199bce378","source":{"kind":"arxiv","id":"2607.15161","version":1},"attestation_state":"computed","paper":{"title":"On-Policy Delta Distillation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.LG","authors_text":"Byeongho Heo, Dongyoon Han, Jaehui Hwang, Sangdoo Yun","submitted_at":"2026-07-16T16:07:19Z","abstract_excerpt":"On-policy distillation is an alternative post-training method in reinforcement learning that alleviates the constraints imposed by reward models by providing token-level supervision from a teacher model. Although on-policy distillation has been studied and applied across various settings, its fundamental design remains underexplored. In this paper, we introduce a new distillation reward, termed the delta signal, instead of directly imitating the teacher's output distribution. The delta signal is defined as the difference between the teacher model and its base model prior to instruction tuning "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2607.15161","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2026-07-16T16:07:19Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"c9803c4df7ba3c5909eb7bf13f75ae636b2fca6b4287ad7e95388c421afb041b","abstract_canon_sha256":"56232899593f8cf3402f3acf8a13d89301b3ca89d144c80e2562edb94d1d2b9f"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-17T01:22:12.351630Z","signature_b64":"Uz4fzoe+FDS1s2PidAQFJUllVvcXKPuL6pD38xZc9yAlrIXCaiuLEtX4EAvrh6FF0g6WbCEpCJRGGaOarY2iAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"787c43f25a60a834b3ef88ad7a5c5ac84f88c4794789fa85bb8028e199bce378","last_reissued_at":"2026-07-17T01:22:12.350789Z","signature_status":"signed_v1","first_computed_at":"2026-07-17T01:22:12.350789Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"On-Policy Delta Distillation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.LG","authors_text":"Byeongho Heo, Dongyoon Han, Jaehui Hwang, Sangdoo Yun","submitted_at":"2026-07-16T16:07:19Z","abstract_excerpt":"On-policy distillation is an alternative post-training method in reinforcement learning that alleviates the constraints imposed by reward models by providing token-level supervision from a teacher model. Although on-policy distillation has been studied and applied across various settings, its fundamental design remains underexplored. In this paper, we introduce a new distillation reward, termed the delta signal, instead of directly imitating the teacher's output distribution. The delta signal is defined as the difference between the teacher model and its base model prior to instruction tuning "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2607.15161","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2607.15161/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2607.15161","created_at":"2026-07-17T01:22:12.351221+00:00"},{"alias_kind":"arxiv_version","alias_value":"2607.15161v1","created_at":"2026-07-17T01:22:12.351221+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2607.15161","created_at":"2026-07-17T01:22:12.351221+00:00"},{"alias_kind":"pith_short_12","alias_value":"PB6EH4S2MCUD","created_at":"2026-07-17T01:22:12.351221+00:00"},{"alias_kind":"pith_short_16","alias_value":"PB6EH4S2MCUDJM7P","created_at":"2026-07-17T01:22:12.351221+00:00"},{"alias_kind":"pith_short_8","alias_value":"PB6EH4S2","created_at":"2026-07-17T01:22:12.351221+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/PB6EH4S2MCUDJM7PRCWXUXC2ZB","json":"https://pith.science/pith/PB6EH4S2MCUDJM7PRCWXUXC2ZB.json","graph_json":"https://pith.science/api/pith-number/PB6EH4S2MCUDJM7PRCWXUXC2ZB/graph.json","events_json":"https://pith.science/api/pith-number/PB6EH4S2MCUDJM7PRCWXUXC2ZB/events.json","paper":"https://pith.science/paper/PB6EH4S2"},"agent_actions":{"view_html":"https://pith.science/pith/PB6EH4S2MCUDJM7PRCWXUXC2ZB","download_json":"https://pith.science/pith/PB6EH4S2MCUDJM7PRCWXUXC2ZB.json","view_paper":"https://pith.science/paper/PB6EH4S2","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2607.15161&json=true","fetch_graph":"https://pith.science/api/pith-number/PB6EH4S2MCUDJM7PRCWXUXC2ZB/graph.json","fetch_events":"https://pith.science/api/pith-number/PB6EH4S2MCUDJM7PRCWXUXC2ZB/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/PB6EH4S2MCUDJM7PRCWXUXC2ZB/action/timestamp_anchor","attest_storage":"https://pith.science/pith/PB6EH4S2MCUDJM7PRCWXUXC2ZB/action/storage_attestation","attest_author":"https://pith.science/pith/PB6EH4S2MCUDJM7PRCWXUXC2ZB/action/author_attestation","sign_citation":"https://pith.science/pith/PB6EH4S2MCUDJM7PRCWXUXC2ZB/action/citation_signature","submit_replication":"https://pith.science/pith/PB6EH4S2MCUDJM7PRCWXUXC2ZB/action/replication_record"}},"created_at":"2026-07-17T01:22:12.351221+00:00","updated_at":"2026-07-17T01:22:12.351221+00:00"}