{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:KBRAPRFIXBG5TNY2K55DNOGJNU","short_pith_number":"pith:KBRAPRFI","schema_version":"1.0","canonical_sha256":"506207c4a8b84dd9b71a577a36b8c96d2c18f5d85d8a241b1a1b1412e0c48a60","source":{"kind":"arxiv","id":"2310.09144","version":1},"attestation_state":"computed","paper":{"title":"Goodhart's Law in Reinforcement Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Charlie Griffin, Jacek Karwowski, Joar Skalse, Klaus Kiendlhofer, Oliver Hayman, Xingjian Bai","submitted_at":"2023-10-13T14:35:59Z","abstract_excerpt":"Implementing a reward function that perfectly captures a complex task in the real world is impractical. As a result, it is often appropriate to think of the reward function as a proxy for the true objective rather than as its definition. We study this phenomenon through the lens of Goodhart's law, which predicts that increasing optimisation of an imperfect proxy beyond some critical point decreases performance on the true objective. First, we propose a way to quantify the magnitude of this effect and show empirically that optimising an imperfect proxy reward often leads to the behaviour predic"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2310.09144","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2023-10-13T14:35:59Z","cross_cats_sorted":[],"title_canon_sha256":"882d0cb473a8937587be704c7f5e895ec9fffe3b2b5b4c623288a718636d9b41","abstract_canon_sha256":"b733e1cc1fbf8be7daf3917c6ebed6330f639c0ce4b53136209df375f065b05a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:00:40.444998Z","signature_b64":"zJqNhnvr6BWFvaIMyxn/j+8ETd0YZdecsJIh2AXQBszZHk6Da1m7QwVVoGEwJYYyVrU0XaV+oIWpuSp5meXbDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"506207c4a8b84dd9b71a577a36b8c96d2c18f5d85d8a241b1a1b1412e0c48a60","last_reissued_at":"2026-07-05T07:00:40.444584Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:00:40.444584Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Goodhart's Law in Reinforcement Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Charlie Griffin, Jacek Karwowski, Joar Skalse, Klaus Kiendlhofer, Oliver Hayman, Xingjian Bai","submitted_at":"2023-10-13T14:35:59Z","abstract_excerpt":"Implementing a reward function that perfectly captures a complex task in the real world is impractical. As a result, it is often appropriate to think of the reward function as a proxy for the true objective rather than as its definition. We study this phenomenon through the lens of Goodhart's law, which predicts that increasing optimisation of an imperfect proxy beyond some critical point decreases performance on the true objective. First, we propose a way to quantify the magnitude of this effect and show empirically that optimising an imperfect proxy reward often leads to the behaviour predic"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2310.09144","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2310.09144/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2310.09144","created_at":"2026-07-05T07:00:40.444640+00:00"},{"alias_kind":"arxiv_version","alias_value":"2310.09144v1","created_at":"2026-07-05T07:00:40.444640+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2310.09144","created_at":"2026-07-05T07:00:40.444640+00:00"},{"alias_kind":"pith_short_12","alias_value":"KBRAPRFIXBG5","created_at":"2026-07-05T07:00:40.444640+00:00"},{"alias_kind":"pith_short_16","alias_value":"KBRAPRFIXBG5TNY2","created_at":"2026-07-05T07:00:40.444640+00:00"},{"alias_kind":"pith_short_8","alias_value":"KBRAPRFI","created_at":"2026-07-05T07:00:40.444640+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":7,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.09711","citing_title":"Proxy Reward Internalization and Mechanistic Exploitation: A Learned Precursor to Reward Hacking and Its Generalization","ref_index":267,"is_internal_anchor":false},{"citing_arxiv_id":"2503.21849","citing_title":"Selection of the fittest or selection of the luckiest: the emergence of Goodhart's law in evolution","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2602.13934","citing_title":"Why Code, Why Now: An Information-Theoretic Perspective on the Limits of Machine Learning","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2604.08805","citing_title":"Building Better Environments for Autonomous Cyber Defence","ref_index":36,"is_internal_anchor":false},{"citing_arxiv_id":"2604.07343","citing_title":"Personalized RewardBench: Evaluating Reward Models with Human Aligned Personalization","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2604.13602","citing_title":"Reward Hacking in the Era of Large Models: Mechanisms, Emergent Misalignment, Challenges","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2604.14910","citing_title":"Reward-Aware Trajectory Shaping for Few-step Visual Generation","ref_index":10,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/KBRAPRFIXBG5TNY2K55DNOGJNU","json":"https://pith.science/pith/KBRAPRFIXBG5TNY2K55DNOGJNU.json","graph_json":"https://pith.science/api/pith-number/KBRAPRFIXBG5TNY2K55DNOGJNU/graph.json","events_json":"https://pith.science/api/pith-number/KBRAPRFIXBG5TNY2K55DNOGJNU/events.json","paper":"https://pith.science/paper/KBRAPRFI"},"agent_actions":{"view_html":"https://pith.science/pith/KBRAPRFIXBG5TNY2K55DNOGJNU","download_json":"https://pith.science/pith/KBRAPRFIXBG5TNY2K55DNOGJNU.json","view_paper":"https://pith.science/paper/KBRAPRFI","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2310.09144&json=true","fetch_graph":"https://pith.science/api/pith-number/KBRAPRFIXBG5TNY2K55DNOGJNU/graph.json","fetch_events":"https://pith.science/api/pith-number/KBRAPRFIXBG5TNY2K55DNOGJNU/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/KBRAPRFIXBG5TNY2K55DNOGJNU/action/timestamp_anchor","attest_storage":"https://pith.science/pith/KBRAPRFIXBG5TNY2K55DNOGJNU/action/storage_attestation","attest_author":"https://pith.science/pith/KBRAPRFIXBG5TNY2K55DNOGJNU/action/author_attestation","sign_citation":"https://pith.science/pith/KBRAPRFIXBG5TNY2K55DNOGJNU/action/citation_signature","submit_replication":"https://pith.science/pith/KBRAPRFIXBG5TNY2K55DNOGJNU/action/replication_record"}},"created_at":"2026-07-05T07:00:40.444640+00:00","updated_at":"2026-07-05T07:00:40.444640+00:00"}