{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:JPV2REKRDDKEMZ5ALECYX6O33I","short_pith_number":"pith:JPV2REKR","schema_version":"1.0","canonical_sha256":"4beba8915118d44667a059058bf9dbda1ec9dc3efc01ef1ef87db701e3b26ef8","source":{"kind":"arxiv","id":"2501.08617","version":3},"attestation_state":"computed","paper":{"title":"RLHS: Mitigating Misalignment in RLHF with Hindsight Simulation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Haimin Hu, Jaime Fern\\'andez Fisac, Kaiqu Liang, Ryan Liu, Thomas L. Griffiths","submitted_at":"2025-01-15T06:33:15Z","abstract_excerpt":"While Reinforcement Learning from Human Feedback (RLHF) has shown promise in aligning generative AI, we present empirical evidence that it can also cause severe, systematic misalignment. We hypothesize that this stems from evaluator feedback depending on downstream outcome predictions (foresight) that can be influenced by the AI's output, inducing Goodhart's law dynamics. We present a theoretical analysis showing that conditioning evaluator feedback on downstream observations (hindsight) inhibits this effect by decoupling the alignment signal from potentially compromised predictions--crucially"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2501.08617","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2025-01-15T06:33:15Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"4f112f0207ec733cb1f093c7b01a6fb3fa7358a5850c74055ceb60d46752083c","abstract_canon_sha256":"df0f822d09330ebe0ce2cd13d1b466abbfda52bf14947064f702a32bb9136caf"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:18:41.171772Z","signature_b64":"Cc0SsPsAk3PkPXH3pcFQFwnB+1FqJeK3EkbtQj+DBnRwsREP9J+e1C5Xkla22+AkMD98vPN+10Xs89hQOGk/AQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"4beba8915118d44667a059058bf9dbda1ec9dc3efc01ef1ef87db701e3b26ef8","last_reissued_at":"2026-07-05T11:18:41.171327Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:18:41.171327Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"RLHS: Mitigating Misalignment in RLHF with Hindsight Simulation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Haimin Hu, Jaime Fern\\'andez Fisac, Kaiqu Liang, Ryan Liu, Thomas L. Griffiths","submitted_at":"2025-01-15T06:33:15Z","abstract_excerpt":"While Reinforcement Learning from Human Feedback (RLHF) has shown promise in aligning generative AI, we present empirical evidence that it can also cause severe, systematic misalignment. We hypothesize that this stems from evaluator feedback depending on downstream outcome predictions (foresight) that can be influenced by the AI's output, inducing Goodhart's law dynamics. We present a theoretical analysis showing that conditioning evaluator feedback on downstream observations (hindsight) inhibits this effect by decoupling the alignment signal from potentially compromised predictions--crucially"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2501.08617","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2501.08617/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2501.08617","created_at":"2026-07-05T11:18:41.171381+00:00"},{"alias_kind":"arxiv_version","alias_value":"2501.08617v3","created_at":"2026-07-05T11:18:41.171381+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2501.08617","created_at":"2026-07-05T11:18:41.171381+00:00"},{"alias_kind":"pith_short_12","alias_value":"JPV2REKRDDKE","created_at":"2026-07-05T11:18:41.171381+00:00"},{"alias_kind":"pith_short_16","alias_value":"JPV2REKRDDKEMZ5A","created_at":"2026-07-05T11:18:41.171381+00:00"},{"alias_kind":"pith_short_8","alias_value":"JPV2REKR","created_at":"2026-07-05T11:18:41.171381+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2510.13727","citing_title":"From Refusal to Recovery: A Control-Theoretic Approach to Generative AI Guardrails","ref_index":84,"is_internal_anchor":false},{"citing_arxiv_id":"2604.08525","citing_title":"Ads in AI Chatbots? An Analysis of How Large Language Models Navigate Conflicts of Interest","ref_index":61,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/JPV2REKRDDKEMZ5ALECYX6O33I","json":"https://pith.science/pith/JPV2REKRDDKEMZ5ALECYX6O33I.json","graph_json":"https://pith.science/api/pith-number/JPV2REKRDDKEMZ5ALECYX6O33I/graph.json","events_json":"https://pith.science/api/pith-number/JPV2REKRDDKEMZ5ALECYX6O33I/events.json","paper":"https://pith.science/paper/JPV2REKR"},"agent_actions":{"view_html":"https://pith.science/pith/JPV2REKRDDKEMZ5ALECYX6O33I","download_json":"https://pith.science/pith/JPV2REKRDDKEMZ5ALECYX6O33I.json","view_paper":"https://pith.science/paper/JPV2REKR","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2501.08617&json=true","fetch_graph":"https://pith.science/api/pith-number/JPV2REKRDDKEMZ5ALECYX6O33I/graph.json","fetch_events":"https://pith.science/api/pith-number/JPV2REKRDDKEMZ5ALECYX6O33I/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/JPV2REKRDDKEMZ5ALECYX6O33I/action/timestamp_anchor","attest_storage":"https://pith.science/pith/JPV2REKRDDKEMZ5ALECYX6O33I/action/storage_attestation","attest_author":"https://pith.science/pith/JPV2REKRDDKEMZ5ALECYX6O33I/action/author_attestation","sign_citation":"https://pith.science/pith/JPV2REKRDDKEMZ5ALECYX6O33I/action/citation_signature","submit_replication":"https://pith.science/pith/JPV2REKRDDKEMZ5ALECYX6O33I/action/replication_record"}},"created_at":"2026-07-05T11:18:41.171381+00:00","updated_at":"2026-07-05T11:18:41.171381+00:00"}