{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:PFH2UJZ2LWFG3E5CDMSK4HXVK7","short_pith_number":"pith:PFH2UJZ2","schema_version":"1.0","canonical_sha256":"794faa273a5d8a6d93a21b24ae1ef557eb88b84806d36ea9a52f91eb54170d0c","source":{"kind":"arxiv","id":"2406.18346","version":1},"attestation_state":"computed","paper":{"title":"AI Alignment through Reinforcement Learning from Human Feedback? Contradictions and Limitations","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Adam Dahlgren Lindstr\\\"om, Dimitri Coelho Mollo, \\'I\\~nigo Mart\\'inez de Rituerto de Troya, Lea Krause, Leila Methnani, Petter Ericson, Roel Dobbe","submitted_at":"2024-06-26T13:42:13Z","abstract_excerpt":"This paper critically evaluates the attempts to align Artificial Intelligence (AI) systems, especially Large Language Models (LLMs), with human values and intentions through Reinforcement Learning from Feedback (RLxF) methods, involving either human feedback (RLHF) or AI feedback (RLAIF). Specifically, we show the shortcomings of the broadly pursued alignment goals of honesty, harmlessness, and helpfulness. Through a multidisciplinary sociotechnical critique, we examine both the theoretical underpinnings and practical implementations of RLxF techniques, revealing significant limitations in the"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2406.18346","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-sa/4.0/","primary_cat":"cs.AI","submitted_at":"2024-06-26T13:42:13Z","cross_cats_sorted":[],"title_canon_sha256":"e296c971cd962a5709f49eccc22f828a0e2ee97c918c1f118bbff1c42a8d2ac6","abstract_canon_sha256":"bbb1a3b2cc7eedd538a5d91f71b002c1e49b8cc64840505bbfcc54c224aa026a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:37:08.147859Z","signature_b64":"NEAoJF5oATpoTxk3LJnlkDm9FnSkFW8ty8biIIT8Zf/29LAGC2Ke4wXtKO0DkTYAjLI+Rjgtt2Ln5gzg2Z5xBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"794faa273a5d8a6d93a21b24ae1ef557eb88b84806d36ea9a52f91eb54170d0c","last_reissued_at":"2026-07-05T08:37:08.147281Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:37:08.147281Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"AI Alignment through Reinforcement Learning from Human Feedback? Contradictions and Limitations","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Adam Dahlgren Lindstr\\\"om, Dimitri Coelho Mollo, \\'I\\~nigo Mart\\'inez de Rituerto de Troya, Lea Krause, Leila Methnani, Petter Ericson, Roel Dobbe","submitted_at":"2024-06-26T13:42:13Z","abstract_excerpt":"This paper critically evaluates the attempts to align Artificial Intelligence (AI) systems, especially Large Language Models (LLMs), with human values and intentions through Reinforcement Learning from Feedback (RLxF) methods, involving either human feedback (RLHF) or AI feedback (RLAIF). Specifically, we show the shortcomings of the broadly pursued alignment goals of honesty, harmlessness, and helpfulness. Through a multidisciplinary sociotechnical critique, we examine both the theoretical underpinnings and practical implementations of RLxF techniques, revealing significant limitations in the"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.18346","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.18346/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2406.18346","created_at":"2026-07-05T08:37:08.147352+00:00"},{"alias_kind":"arxiv_version","alias_value":"2406.18346v1","created_at":"2026-07-05T08:37:08.147352+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.18346","created_at":"2026-07-05T08:37:08.147352+00:00"},{"alias_kind":"pith_short_12","alias_value":"PFH2UJZ2LWFG","created_at":"2026-07-05T08:37:08.147352+00:00"},{"alias_kind":"pith_short_16","alias_value":"PFH2UJZ2LWFG3E5C","created_at":"2026-07-05T08:37:08.147352+00:00"},{"alias_kind":"pith_short_8","alias_value":"PFH2UJZ2","created_at":"2026-07-05T08:37:08.147352+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.07605","citing_title":"User identity conditions moral wrongness ratings in non-reasoning large language models","ref_index":18,"is_internal_anchor":true},{"citing_arxiv_id":"2606.06674","citing_title":"What Do People Actually Want From AI? Mapping Preference Plurality","ref_index":58,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17458","citing_title":"ClaHF: A Human Feedback-inspired Reinforcement Learning Framework for Improving Classification Tasks","ref_index":53,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/PFH2UJZ2LWFG3E5CDMSK4HXVK7","json":"https://pith.science/pith/PFH2UJZ2LWFG3E5CDMSK4HXVK7.json","graph_json":"https://pith.science/api/pith-number/PFH2UJZ2LWFG3E5CDMSK4HXVK7/graph.json","events_json":"https://pith.science/api/pith-number/PFH2UJZ2LWFG3E5CDMSK4HXVK7/events.json","paper":"https://pith.science/paper/PFH2UJZ2"},"agent_actions":{"view_html":"https://pith.science/pith/PFH2UJZ2LWFG3E5CDMSK4HXVK7","download_json":"https://pith.science/pith/PFH2UJZ2LWFG3E5CDMSK4HXVK7.json","view_paper":"https://pith.science/paper/PFH2UJZ2","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2406.18346&json=true","fetch_graph":"https://pith.science/api/pith-number/PFH2UJZ2LWFG3E5CDMSK4HXVK7/graph.json","fetch_events":"https://pith.science/api/pith-number/PFH2UJZ2LWFG3E5CDMSK4HXVK7/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/PFH2UJZ2LWFG3E5CDMSK4HXVK7/action/timestamp_anchor","attest_storage":"https://pith.science/pith/PFH2UJZ2LWFG3E5CDMSK4HXVK7/action/storage_attestation","attest_author":"https://pith.science/pith/PFH2UJZ2LWFG3E5CDMSK4HXVK7/action/author_attestation","sign_citation":"https://pith.science/pith/PFH2UJZ2LWFG3E5CDMSK4HXVK7/action/citation_signature","submit_replication":"https://pith.science/pith/PFH2UJZ2LWFG3E5CDMSK4HXVK7/action/replication_record"}},"created_at":"2026-07-05T08:37:08.147352+00:00","updated_at":"2026-07-05T08:37:08.147352+00:00"}