{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:UNMQCIPZ52V2TPOBE4RNKBVXWJ","short_pith_number":"pith:UNMQCIPZ","schema_version":"1.0","canonical_sha256":"a3590121f9eeaba9bdc12722d506b7b24f719a10b36b0e120534c896237d91ac","source":{"kind":"arxiv","id":"2204.06601","version":4},"attestation_state":"computed","paper":{"title":"Causal Confusion and Reward Misidentification in Preference-Based Reward Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.RO"],"primary_cat":"cs.LG","authors_text":"Anca D. Dragan, Daniel S. Brown, Jeremy Tien, Jerry Zhi-Yang He, Zackory Erickson","submitted_at":"2022-04-13T18:41:41Z","abstract_excerpt":"Learning policies via preference-based reward learning is an increasingly popular method for customizing agent behavior, but has been shown anecdotally to be prone to spurious correlations and reward hacking behaviors. While much prior work focuses on causal confusion in reinforcement learning and behavioral cloning, we focus on a systematic study of causal confusion and reward misidentification when learning from preferences. In particular, we perform a series of sensitivity and ablation analyses on several benchmark domains where rewards learned from preferences achieve minimal test error bu"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2204.06601","kind":"arxiv","version":4},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2022-04-13T18:41:41Z","cross_cats_sorted":["cs.RO"],"title_canon_sha256":"7cd5746a8413a57ca00880fc7f1fc83985ea59bf9149d81bb7f1ebe65cbcdd6c","abstract_canon_sha256":"9bf867336581c4e1911fbfb484e0fb7cfd40a02ea1f325ac24cb38b2ccd711fd"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:52:19.462007Z","signature_b64":"NFEIgxYmzPWQ3tzGyMr01H+dnu+YvKWtcS6lpyrTmFnS/ysD1fEfQ+UfYdGUrIiuhWL9vMmggfqK9W1vo+eHCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"a3590121f9eeaba9bdc12722d506b7b24f719a10b36b0e120534c896237d91ac","last_reissued_at":"2026-07-05T05:52:19.461582Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:52:19.461582Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Causal Confusion and Reward Misidentification in Preference-Based Reward Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.RO"],"primary_cat":"cs.LG","authors_text":"Anca D. Dragan, Daniel S. Brown, Jeremy Tien, Jerry Zhi-Yang He, Zackory Erickson","submitted_at":"2022-04-13T18:41:41Z","abstract_excerpt":"Learning policies via preference-based reward learning is an increasingly popular method for customizing agent behavior, but has been shown anecdotally to be prone to spurious correlations and reward hacking behaviors. While much prior work focuses on causal confusion in reinforcement learning and behavioral cloning, we focus on a systematic study of causal confusion and reward misidentification when learning from preferences. In particular, we perform a series of sensitivity and ablation analyses on several benchmark domains where rewards learned from preferences achieve minimal test error bu"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2204.06601","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2204.06601/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2204.06601","created_at":"2026-07-05T05:52:19.461639+00:00"},{"alias_kind":"arxiv_version","alias_value":"2204.06601v4","created_at":"2026-07-05T05:52:19.461639+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2204.06601","created_at":"2026-07-05T05:52:19.461639+00:00"},{"alias_kind":"pith_short_12","alias_value":"UNMQCIPZ52V2","created_at":"2026-07-05T05:52:19.461639+00:00"},{"alias_kind":"pith_short_16","alias_value":"UNMQCIPZ52V2TPOB","created_at":"2026-07-05T05:52:19.461639+00:00"},{"alias_kind":"pith_short_8","alias_value":"UNMQCIPZ","created_at":"2026-07-05T05:52:19.461639+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2509.03403","citing_title":"Beyond Correctness: Harmonizing Process and Outcome Rewards through RL Training","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2605.16339","citing_title":"Preference Instability in Reward Models: Detection and Mitigation via Sparse Autoencoders","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2509.23102","citing_title":"Multiplayer Nash Preference Optimization","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2304.06767","citing_title":"RAFT: Reward rAnked FineTuning for Generative Foundation Model Alignment","ref_index":125,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/UNMQCIPZ52V2TPOBE4RNKBVXWJ","json":"https://pith.science/pith/UNMQCIPZ52V2TPOBE4RNKBVXWJ.json","graph_json":"https://pith.science/api/pith-number/UNMQCIPZ52V2TPOBE4RNKBVXWJ/graph.json","events_json":"https://pith.science/api/pith-number/UNMQCIPZ52V2TPOBE4RNKBVXWJ/events.json","paper":"https://pith.science/paper/UNMQCIPZ"},"agent_actions":{"view_html":"https://pith.science/pith/UNMQCIPZ52V2TPOBE4RNKBVXWJ","download_json":"https://pith.science/pith/UNMQCIPZ52V2TPOBE4RNKBVXWJ.json","view_paper":"https://pith.science/paper/UNMQCIPZ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2204.06601&json=true","fetch_graph":"https://pith.science/api/pith-number/UNMQCIPZ52V2TPOBE4RNKBVXWJ/graph.json","fetch_events":"https://pith.science/api/pith-number/UNMQCIPZ52V2TPOBE4RNKBVXWJ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/UNMQCIPZ52V2TPOBE4RNKBVXWJ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/UNMQCIPZ52V2TPOBE4RNKBVXWJ/action/storage_attestation","attest_author":"https://pith.science/pith/UNMQCIPZ52V2TPOBE4RNKBVXWJ/action/author_attestation","sign_citation":"https://pith.science/pith/UNMQCIPZ52V2TPOBE4RNKBVXWJ/action/citation_signature","submit_replication":"https://pith.science/pith/UNMQCIPZ52V2TPOBE4RNKBVXWJ/action/replication_record"}},"created_at":"2026-07-05T05:52:19.461639+00:00","updated_at":"2026-07-05T05:52:19.461639+00:00"}