{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:3LPH36JOMYMUG76AAR5IZJRKIV","short_pith_number":"pith:3LPH36JO","schema_version":"1.0","canonical_sha256":"dade7df92e6619437fc0047a8ca62a455ea56e1ad951ddf68d4d9773a07c2306","source":{"kind":"arxiv","id":"2403.03185","version":4},"attestation_state":"computed","paper":{"title":"Correlated Proxies: A New Definition and Improved Mitigation for Reward Hacking","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Anca Dragan, Cassidy Laidlaw, Shivam Singhal","submitted_at":"2024-03-05T18:22:15Z","abstract_excerpt":"Because it is difficult to precisely specify complex objectives, reinforcement learning policies are often optimized using proxy reward functions that only approximate the true goal. However, optimizing proxy rewards frequently leads to reward hacking: the optimized reward function ceases to be a good proxy and the resulting policy performs poorly with respect to the unspecified true reward. Principled solutions to reward hacking have been impeded by the lack of a good definition for the problem. To address this gap, we introduce a definition of reward hacking based on the correlation between "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2403.03185","kind":"arxiv","version":4},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-03-05T18:22:15Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"80efa56f65cd8ee73b5ce41f39c4d81f6a6f366927960f7e49b3c07bac55b21f","abstract_canon_sha256":"4fcd0503fb5dbf2ee89de3ec3ef96ec48d24d40046bfcf351612693a305e0685"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:30:19.788512Z","signature_b64":"yaKcKhgeXn/cYEHNTPDakK4qaI5yneUruoY6rYnXKtmK0RnLziEcPCpbssACupdDCXSxVEbZ/3PUmptF5OqxCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"dade7df92e6619437fc0047a8ca62a455ea56e1ad951ddf68d4d9773a07c2306","last_reissued_at":"2026-07-05T10:30:19.787934Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:30:19.787934Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Correlated Proxies: A New Definition and Improved Mitigation for Reward Hacking","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Anca Dragan, Cassidy Laidlaw, Shivam Singhal","submitted_at":"2024-03-05T18:22:15Z","abstract_excerpt":"Because it is difficult to precisely specify complex objectives, reinforcement learning policies are often optimized using proxy reward functions that only approximate the true goal. However, optimizing proxy rewards frequently leads to reward hacking: the optimized reward function ceases to be a good proxy and the resulting policy performs poorly with respect to the unspecified true reward. Principled solutions to reward hacking have been impeded by the lack of a good definition for the problem. To address this gap, we introduce a definition of reward hacking based on the correlation between "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2403.03185","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2403.03185/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2403.03185","created_at":"2026-07-05T10:30:19.787998+00:00"},{"alias_kind":"arxiv_version","alias_value":"2403.03185v4","created_at":"2026-07-05T10:30:19.787998+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2403.03185","created_at":"2026-07-05T10:30:19.787998+00:00"},{"alias_kind":"pith_short_12","alias_value":"3LPH36JOMYMU","created_at":"2026-07-05T10:30:19.787998+00:00"},{"alias_kind":"pith_short_16","alias_value":"3LPH36JOMYMUG76A","created_at":"2026-07-05T10:30:19.787998+00:00"},{"alias_kind":"pith_short_8","alias_value":"3LPH36JO","created_at":"2026-07-05T10:30:19.787998+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":7,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2604.23488","citing_title":"Do Prompt-Elicited Trajectories Reflect Training-Time Reward Hacking? A Systematic Study on Monitoring Training-Time Reward Hacking in Code Generation","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2606.30666","citing_title":"Reframing AGI Confrontation with Off Earth Autonomy","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2605.25189","citing_title":"Directional Alignment Mitigates Reward Hacking in Reinforcement Learning for Language Models","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2605.30719","citing_title":"When are LLMs Sufficient Policy Optimizers for Sequential RL Tasks?","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2605.30719","citing_title":"When are LLMs Sufficient Policy Optimizers for Sequential RL Tasks?","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2604.23488","citing_title":"Do Prompt-Elicited Trajectories Reflect Training-Time Reward Hacking? A Systematic Study on Monitoring Training-Time Reward Hacking in Code Generation","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2604.07484","citing_title":"ConsistRM: Improving Generative Reward Models via Consistency-Aware Self-Training","ref_index":2,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/3LPH36JOMYMUG76AAR5IZJRKIV","json":"https://pith.science/pith/3LPH36JOMYMUG76AAR5IZJRKIV.json","graph_json":"https://pith.science/api/pith-number/3LPH36JOMYMUG76AAR5IZJRKIV/graph.json","events_json":"https://pith.science/api/pith-number/3LPH36JOMYMUG76AAR5IZJRKIV/events.json","paper":"https://pith.science/paper/3LPH36JO"},"agent_actions":{"view_html":"https://pith.science/pith/3LPH36JOMYMUG76AAR5IZJRKIV","download_json":"https://pith.science/pith/3LPH36JOMYMUG76AAR5IZJRKIV.json","view_paper":"https://pith.science/paper/3LPH36JO","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2403.03185&json=true","fetch_graph":"https://pith.science/api/pith-number/3LPH36JOMYMUG76AAR5IZJRKIV/graph.json","fetch_events":"https://pith.science/api/pith-number/3LPH36JOMYMUG76AAR5IZJRKIV/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/3LPH36JOMYMUG76AAR5IZJRKIV/action/timestamp_anchor","attest_storage":"https://pith.science/pith/3LPH36JOMYMUG76AAR5IZJRKIV/action/storage_attestation","attest_author":"https://pith.science/pith/3LPH36JOMYMUG76AAR5IZJRKIV/action/author_attestation","sign_citation":"https://pith.science/pith/3LPH36JOMYMUG76AAR5IZJRKIV/action/citation_signature","submit_replication":"https://pith.science/pith/3LPH36JOMYMUG76AAR5IZJRKIV/action/replication_record"}},"created_at":"2026-07-05T10:30:19.787998+00:00","updated_at":"2026-07-05T10:30:19.787998+00:00"}