{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:FZDSHQFGGPNWHAY6VGRW2DSISR","short_pith_number":"pith:FZDSHQFG","schema_version":"1.0","canonical_sha256":"2e4723c0a633db63831ea9a36d0e489449da2a3f40250fc8a06968120da7a34c","source":{"kind":"arxiv","id":"2412.00967","version":1},"attestation_state":"computed","paper":{"title":"Linear Probe Penalties Reduce LLM Sycophancy","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Henry Papadatos, Rachel Freedman","submitted_at":"2024-12-01T21:11:28Z","abstract_excerpt":"Large language models (LLMs) are often sycophantic, prioritizing agreement with their users over accurate or objective statements. This problematic behavior becomes more pronounced during reinforcement learning from human feedback (RLHF), an LLM fine-tuning stage intended to align model outputs with human values. Instead of increasing accuracy and reliability, the reward model learned from RLHF often rewards sycophancy. We develop a linear probing method to identify and penalize markers of sycophancy within the reward model, producing rewards that discourage sycophantic behavior. Our experimen"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2412.00967","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2024-12-01T21:11:28Z","cross_cats_sorted":[],"title_canon_sha256":"e34d789b78103c8cb602fcf68b112b3240875ae9cfa77b6da8aea66fbc47cdde","abstract_canon_sha256":"697a933aa1137ceab0f378fb1a98c3d5ace67d844f26caf2518d629f672becc1"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:42:51.308611Z","signature_b64":"REFX/fnRMj7/tUtZtmeU8j+Ag5kQGmv83i6uYUK780ysak+/sD9/0e3yRbt5YJZUBmyIKJ9Z+BL4yk7VB5WrBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"2e4723c0a633db63831ea9a36d0e489449da2a3f40250fc8a06968120da7a34c","last_reissued_at":"2026-07-05T09:42:51.308069Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:42:51.308069Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Linear Probe Penalties Reduce LLM Sycophancy","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Henry Papadatos, Rachel Freedman","submitted_at":"2024-12-01T21:11:28Z","abstract_excerpt":"Large language models (LLMs) are often sycophantic, prioritizing agreement with their users over accurate or objective statements. This problematic behavior becomes more pronounced during reinforcement learning from human feedback (RLHF), an LLM fine-tuning stage intended to align model outputs with human values. Instead of increasing accuracy and reliability, the reward model learned from RLHF often rewards sycophancy. We develop a linear probing method to identify and penalize markers of sycophancy within the reward model, producing rewards that discourage sycophantic behavior. Our experimen"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2412.00967","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2412.00967/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2412.00967","created_at":"2026-07-05T09:42:51.308129+00:00"},{"alias_kind":"arxiv_version","alias_value":"2412.00967v1","created_at":"2026-07-05T09:42:51.308129+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2412.00967","created_at":"2026-07-05T09:42:51.308129+00:00"},{"alias_kind":"pith_short_12","alias_value":"FZDSHQFGGPNW","created_at":"2026-07-05T09:42:51.308129+00:00"},{"alias_kind":"pith_short_16","alias_value":"FZDSHQFGGPNWHAY6","created_at":"2026-07-05T09:42:51.308129+00:00"},{"alias_kind":"pith_short_8","alias_value":"FZDSHQFG","created_at":"2026-07-05T09:42:51.308129+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.09068","citing_title":"Emergent Misalignment Can Be Induced by Sycophancy and Reversed via Alignment Gating","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01302","citing_title":"Beyond Semantic Relevance: Counterfactual Risk Minimization for Robust Retrieval-Augmented Generation","ref_index":62,"is_internal_anchor":false},{"citing_arxiv_id":"2604.05279","citing_title":"Pressure, What Pressure? Sycophancy Disentanglement in Language Models via Reward Decomposition","ref_index":6,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/FZDSHQFGGPNWHAY6VGRW2DSISR","json":"https://pith.science/pith/FZDSHQFGGPNWHAY6VGRW2DSISR.json","graph_json":"https://pith.science/api/pith-number/FZDSHQFGGPNWHAY6VGRW2DSISR/graph.json","events_json":"https://pith.science/api/pith-number/FZDSHQFGGPNWHAY6VGRW2DSISR/events.json","paper":"https://pith.science/paper/FZDSHQFG"},"agent_actions":{"view_html":"https://pith.science/pith/FZDSHQFGGPNWHAY6VGRW2DSISR","download_json":"https://pith.science/pith/FZDSHQFGGPNWHAY6VGRW2DSISR.json","view_paper":"https://pith.science/paper/FZDSHQFG","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2412.00967&json=true","fetch_graph":"https://pith.science/api/pith-number/FZDSHQFGGPNWHAY6VGRW2DSISR/graph.json","fetch_events":"https://pith.science/api/pith-number/FZDSHQFGGPNWHAY6VGRW2DSISR/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/FZDSHQFGGPNWHAY6VGRW2DSISR/action/timestamp_anchor","attest_storage":"https://pith.science/pith/FZDSHQFGGPNWHAY6VGRW2DSISR/action/storage_attestation","attest_author":"https://pith.science/pith/FZDSHQFGGPNWHAY6VGRW2DSISR/action/author_attestation","sign_citation":"https://pith.science/pith/FZDSHQFGGPNWHAY6VGRW2DSISR/action/citation_signature","submit_replication":"https://pith.science/pith/FZDSHQFGGPNWHAY6VGRW2DSISR/action/replication_record"}},"created_at":"2026-07-05T09:42:51.308129+00:00","updated_at":"2026-07-05T09:42:51.308129+00:00"}