{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:R6Q24PJYBYCWQBK4I6SQKBWFYW","short_pith_number":"pith:R6Q24PJY","schema_version":"1.0","canonical_sha256":"8fa1ae3d380e0568055c47a50506c5c5bac9ab7ddb696279c68f090c8005d37d","source":{"kind":"arxiv","id":"2311.14455","version":4},"attestation_state":"computed","paper":{"title":"Universal Jailbreak Backdoors from Poisoned Human Feedback","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL","cs.CR","cs.LG"],"primary_cat":"cs.AI","authors_text":"Florian Tram\\`er, Javier Rando","submitted_at":"2023-11-24T13:09:34Z","abstract_excerpt":"Reinforcement Learning from Human Feedback (RLHF) is used to align large language models to produce helpful and harmless responses. Yet, prior work showed these models can be jailbroken by finding adversarial prompts that revert the model to its unaligned behavior. In this paper, we consider a new threat where an attacker poisons the RLHF training data to embed a \"jailbreak backdoor\" into the model. The backdoor embeds a trigger word into the model that acts like a universal \"sudo command\": adding the trigger word to any prompt enables harmful responses without the need to search for an advers"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2311.14455","kind":"arxiv","version":4},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2023-11-24T13:09:34Z","cross_cats_sorted":["cs.CL","cs.CR","cs.LG"],"title_canon_sha256":"72b308ab6315f53f8ed7871b0a61a3cd2da78fe2816b9fbf23cdb189525b0abf","abstract_canon_sha256":"793287c1bf6674a1611153bf9f7904e6883b66cad10e78164b4b3ffbaf742b96"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:13:10.212658Z","signature_b64":"9P+bPSK2N7FMAcHNaI9OUMSwVctf+qT0cR+ynvae0kFta1XJ3rCb8iVkKC78pMnQnFWZwiJtKOcUnW+Rh4Q8Dw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"8fa1ae3d380e0568055c47a50506c5c5bac9ab7ddb696279c68f090c8005d37d","last_reissued_at":"2026-07-05T08:13:10.212252Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:13:10.212252Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Universal Jailbreak Backdoors from Poisoned Human Feedback","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL","cs.CR","cs.LG"],"primary_cat":"cs.AI","authors_text":"Florian Tram\\`er, Javier Rando","submitted_at":"2023-11-24T13:09:34Z","abstract_excerpt":"Reinforcement Learning from Human Feedback (RLHF) is used to align large language models to produce helpful and harmless responses. Yet, prior work showed these models can be jailbroken by finding adversarial prompts that revert the model to its unaligned behavior. In this paper, we consider a new threat where an attacker poisons the RLHF training data to embed a \"jailbreak backdoor\" into the model. The backdoor embeds a trigger word into the model that acts like a universal \"sudo command\": adding the trigger word to any prompt enables harmful responses without the need to search for an advers"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2311.14455","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2311.14455/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2311.14455","created_at":"2026-07-05T08:13:10.212302+00:00"},{"alias_kind":"arxiv_version","alias_value":"2311.14455v4","created_at":"2026-07-05T08:13:10.212302+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2311.14455","created_at":"2026-07-05T08:13:10.212302+00:00"},{"alias_kind":"pith_short_12","alias_value":"R6Q24PJYBYCW","created_at":"2026-07-05T08:13:10.212302+00:00"},{"alias_kind":"pith_short_16","alias_value":"R6Q24PJYBYCWQBK4","created_at":"2026-07-05T08:13:10.212302+00:00"},{"alias_kind":"pith_short_8","alias_value":"R6Q24PJY","created_at":"2026-07-05T08:13:10.212302+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":10,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.09411","citing_title":"Now You (Still) See Me: Detecting Evasive Steganographic Payloads in LLMs","ref_index":41,"is_internal_anchor":false},{"citing_arxiv_id":"2605.15152","citing_title":"Widening the Gap: Exploiting LLM Quantization via Outlier Injection","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2504.20984","citing_title":"ACE: A Security Architecture for LLM-Integrated App Systems","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2605.16339","citing_title":"Preference Instability in Reward Models: Detection and Mitigation via Sparse Autoencoders","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2605.16471","citing_title":"From AI-Generated Content to Agentic Action: Security and Safety Threats in Generative AI","ref_index":107,"is_internal_anchor":false},{"citing_arxiv_id":"2507.02850","citing_title":"LLM Hypnosis: Exploiting User Feedback for Unauthorized Knowledge Injection to All Users","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2603.27517","citing_title":"A Security Analysis of the OpenClaw AI Agent Framework","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2603.27517","citing_title":"A Security Analysis of the OpenClaw AI Agent Framework","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09397","citing_title":"BadDLM: Backdooring Diffusion Language Models with Diverse Targets","ref_index":42,"is_internal_anchor":false},{"citing_arxiv_id":"2605.02495","citing_title":"Efficient Preference Poisoning Attack on Offline RLHF","ref_index":44,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/R6Q24PJYBYCWQBK4I6SQKBWFYW","json":"https://pith.science/pith/R6Q24PJYBYCWQBK4I6SQKBWFYW.json","graph_json":"https://pith.science/api/pith-number/R6Q24PJYBYCWQBK4I6SQKBWFYW/graph.json","events_json":"https://pith.science/api/pith-number/R6Q24PJYBYCWQBK4I6SQKBWFYW/events.json","paper":"https://pith.science/paper/R6Q24PJY"},"agent_actions":{"view_html":"https://pith.science/pith/R6Q24PJYBYCWQBK4I6SQKBWFYW","download_json":"https://pith.science/pith/R6Q24PJYBYCWQBK4I6SQKBWFYW.json","view_paper":"https://pith.science/paper/R6Q24PJY","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2311.14455&json=true","fetch_graph":"https://pith.science/api/pith-number/R6Q24PJYBYCWQBK4I6SQKBWFYW/graph.json","fetch_events":"https://pith.science/api/pith-number/R6Q24PJYBYCWQBK4I6SQKBWFYW/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/R6Q24PJYBYCWQBK4I6SQKBWFYW/action/timestamp_anchor","attest_storage":"https://pith.science/pith/R6Q24PJYBYCWQBK4I6SQKBWFYW/action/storage_attestation","attest_author":"https://pith.science/pith/R6Q24PJYBYCWQBK4I6SQKBWFYW/action/author_attestation","sign_citation":"https://pith.science/pith/R6Q24PJYBYCWQBK4I6SQKBWFYW/action/citation_signature","submit_replication":"https://pith.science/pith/R6Q24PJYBYCWQBK4I6SQKBWFYW/action/replication_record"}},"created_at":"2026-07-05T08:13:10.212302+00:00","updated_at":"2026-07-05T08:13:10.212302+00:00"}