{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:JWHOAL6WPD63RR35EKP5SBFVWX","short_pith_number":"pith:JWHOAL6W","schema_version":"1.0","canonical_sha256":"4d8ee02fd678fdb8c77d229fd904b5b5d5fafe1c5c28c3a615342d3770cd88d0","source":{"kind":"arxiv","id":"2311.05553","version":3},"attestation_state":"computed","paper":{"title":"Removing RLHF Protections in GPT-4 via Fine-Tuning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Akul Gupta, Daniel Kang, Qiusi Zhan, Richard Fang, Rohan Bindu, Tatsunori Hashimoto","submitted_at":"2023-11-09T17:54:59Z","abstract_excerpt":"As large language models (LLMs) have increased in their capabilities, so does their potential for dual use. To reduce harmful outputs, produces and vendors of LLMs have used reinforcement learning with human feedback (RLHF). In tandem, LLM vendors have been increasingly enabling fine-tuning of their most powerful models. However, concurrent work has shown that fine-tuning can remove RLHF protections. We may expect that the most powerful models currently available (GPT-4) are less susceptible to fine-tuning attacks. In this work, we show the contrary: fine-tuning allows attackers to remove RLHF"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2311.05553","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2023-11-09T17:54:59Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"118b451a412d4e3770209deb25e54b255112c631edbe893c6eaca194664a7692","abstract_canon_sha256":"9405afe940e9c87c35472f4ad1d99d1eb8856f607393448abf0dfb43bde45c42"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:05:04.876918Z","signature_b64":"5UtcLDDlGry8d1RTrj3zV/FPofGuakUGnqQb964vXZf66b0VwKsk5SdlANAOKthWb8Qqu9cfJpjdFHDdNK1hDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"4d8ee02fd678fdb8c77d229fd904b5b5d5fafe1c5c28c3a615342d3770cd88d0","last_reissued_at":"2026-07-05T08:05:04.876431Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:05:04.876431Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Removing RLHF Protections in GPT-4 via Fine-Tuning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Akul Gupta, Daniel Kang, Qiusi Zhan, Richard Fang, Rohan Bindu, Tatsunori Hashimoto","submitted_at":"2023-11-09T17:54:59Z","abstract_excerpt":"As large language models (LLMs) have increased in their capabilities, so does their potential for dual use. To reduce harmful outputs, produces and vendors of LLMs have used reinforcement learning with human feedback (RLHF). In tandem, LLM vendors have been increasingly enabling fine-tuning of their most powerful models. However, concurrent work has shown that fine-tuning can remove RLHF protections. We may expect that the most powerful models currently available (GPT-4) are less susceptible to fine-tuning attacks. In this work, we show the contrary: fine-tuning allows attackers to remove RLHF"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2311.05553","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2311.05553/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2311.05553","created_at":"2026-07-05T08:05:04.876489+00:00"},{"alias_kind":"arxiv_version","alias_value":"2311.05553v3","created_at":"2026-07-05T08:05:04.876489+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2311.05553","created_at":"2026-07-05T08:05:04.876489+00:00"},{"alias_kind":"pith_short_12","alias_value":"JWHOAL6WPD63","created_at":"2026-07-05T08:05:04.876489+00:00"},{"alias_kind":"pith_short_16","alias_value":"JWHOAL6WPD63RR35","created_at":"2026-07-05T08:05:04.876489+00:00"},{"alias_kind":"pith_short_8","alias_value":"JWHOAL6W","created_at":"2026-07-05T08:05:04.876489+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":10,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.19890","citing_title":"Open Weight AI Models Require Proportional Evaluation Approaches","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2409.18169","citing_title":"Harmful Fine-tuning Attacks and Defenses for Large Language Models: A Survey","ref_index":177,"is_internal_anchor":false},{"citing_arxiv_id":"2404.08144","citing_title":"LLM Agents can Autonomously Exploit One-day Vulnerabilities","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2402.10260","citing_title":"A StrongREJECT for Empty Jailbreaks","ref_index":42,"is_internal_anchor":false},{"citing_arxiv_id":"2602.08813","citing_title":"Robust Policy Optimization to Prevent Catastrophic Forgetting","ref_index":56,"is_internal_anchor":false},{"citing_arxiv_id":"2407.04295","citing_title":"Jailbreak Attacks and Defenses Against Large Language Models: A Survey","ref_index":111,"is_internal_anchor":false},{"citing_arxiv_id":"2406.11717","citing_title":"Refusal in Language Models Is Mediated by a Single Direction","ref_index":205,"is_internal_anchor":false},{"citing_arxiv_id":"2605.04992","citing_title":"You Snooze, You Lose: Automatic Safety Alignment Restoration through Neural Weight Translation","ref_index":146,"is_internal_anchor":false},{"citing_arxiv_id":"2604.14990","citing_title":"The Possibility of Artificial Intelligence Becoming a Subject and the Alignment Problem","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2604.17396","citing_title":"Representation-Guided Parameter-Efficient LLM Unlearning","ref_index":38,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/JWHOAL6WPD63RR35EKP5SBFVWX","json":"https://pith.science/pith/JWHOAL6WPD63RR35EKP5SBFVWX.json","graph_json":"https://pith.science/api/pith-number/JWHOAL6WPD63RR35EKP5SBFVWX/graph.json","events_json":"https://pith.science/api/pith-number/JWHOAL6WPD63RR35EKP5SBFVWX/events.json","paper":"https://pith.science/paper/JWHOAL6W"},"agent_actions":{"view_html":"https://pith.science/pith/JWHOAL6WPD63RR35EKP5SBFVWX","download_json":"https://pith.science/pith/JWHOAL6WPD63RR35EKP5SBFVWX.json","view_paper":"https://pith.science/paper/JWHOAL6W","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2311.05553&json=true","fetch_graph":"https://pith.science/api/pith-number/JWHOAL6WPD63RR35EKP5SBFVWX/graph.json","fetch_events":"https://pith.science/api/pith-number/JWHOAL6WPD63RR35EKP5SBFVWX/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/JWHOAL6WPD63RR35EKP5SBFVWX/action/timestamp_anchor","attest_storage":"https://pith.science/pith/JWHOAL6WPD63RR35EKP5SBFVWX/action/storage_attestation","attest_author":"https://pith.science/pith/JWHOAL6WPD63RR35EKP5SBFVWX/action/author_attestation","sign_citation":"https://pith.science/pith/JWHOAL6WPD63RR35EKP5SBFVWX/action/citation_signature","submit_replication":"https://pith.science/pith/JWHOAL6WPD63RR35EKP5SBFVWX/action/replication_record"}},"created_at":"2026-07-05T08:05:04.876489+00:00","updated_at":"2026-07-05T08:05:04.876489+00:00"}