{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:OTA2I4NJZCJ5VTPATZ4RIQKDKQ","short_pith_number":"pith:OTA2I4NJ","schema_version":"1.0","canonical_sha256":"74c1a471a9c893dacde09e79144143542eb0f941067c13867ca14ebcfbc75384","source":{"kind":"arxiv","id":"2309.01446","version":4},"attestation_state":"computed","paper":{"title":"Open Sesame! Universal Black Box Jailbreaking of Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CV","cs.NE"],"primary_cat":"cs.CL","authors_text":"Moshe Sipper, Raz Lapid, Ron Langberg","submitted_at":"2023-09-04T08:54:20Z","abstract_excerpt":"Large language models (LLMs), designed to provide helpful and safe responses, often rely on alignment techniques to align with user intent and social guidelines. Unfortunately, this alignment can be exploited by malicious actors seeking to manipulate an LLM's outputs for unintended purposes. In this paper we introduce a novel approach that employs a genetic algorithm (GA) to manipulate LLMs when model architecture and parameters are inaccessible. The GA attack works by optimizing a universal adversarial prompt that -- when combined with a user's query -- disrupts the attacked model's alignment"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2309.01446","kind":"arxiv","version":4},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2023-09-04T08:54:20Z","cross_cats_sorted":["cs.CV","cs.NE"],"title_canon_sha256":"982999b78d4c52fd6baeceb6496e211ca2992489410463b3afa8ee954fa154df","abstract_canon_sha256":"70ca0f360eefb9ec008195351fa2b95f0e97c663faeaf65d971e45dfe821a186"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:51:43.353311Z","signature_b64":"1ybkDEWL708xRaW33ARbb0Ri8JWrcDpKXI4VymMNKZlUM7iliyg8qibQTl2+cY4MMOdt5t7wFowDFzTJXDDmAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"74c1a471a9c893dacde09e79144143542eb0f941067c13867ca14ebcfbc75384","last_reissued_at":"2026-07-05T08:51:43.352840Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:51:43.352840Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Open Sesame! Universal Black Box Jailbreaking of Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CV","cs.NE"],"primary_cat":"cs.CL","authors_text":"Moshe Sipper, Raz Lapid, Ron Langberg","submitted_at":"2023-09-04T08:54:20Z","abstract_excerpt":"Large language models (LLMs), designed to provide helpful and safe responses, often rely on alignment techniques to align with user intent and social guidelines. Unfortunately, this alignment can be exploited by malicious actors seeking to manipulate an LLM's outputs for unintended purposes. In this paper we introduce a novel approach that employs a genetic algorithm (GA) to manipulate LLMs when model architecture and parameters are inaccessible. The GA attack works by optimizing a universal adversarial prompt that -- when combined with a user's query -- disrupts the attacked model's alignment"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2309.01446","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2309.01446/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2309.01446","created_at":"2026-07-05T08:51:43.352896+00:00"},{"alias_kind":"arxiv_version","alias_value":"2309.01446v4","created_at":"2026-07-05T08:51:43.352896+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2309.01446","created_at":"2026-07-05T08:51:43.352896+00:00"},{"alias_kind":"pith_short_12","alias_value":"OTA2I4NJZCJ5","created_at":"2026-07-05T08:51:43.352896+00:00"},{"alias_kind":"pith_short_16","alias_value":"OTA2I4NJZCJ5VTPA","created_at":"2026-07-05T08:51:43.352896+00:00"},{"alias_kind":"pith_short_8","alias_value":"OTA2I4NJ","created_at":"2026-07-05T08:51:43.352896+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":12,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2511.12710","citing_title":"Evolve the Method, Not the Prompts: Evolutionary Synthesis of Jailbreak Attacks on LLMs","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2506.12382","citing_title":"Exploring the Secondary Risks of Large Language Models","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2506.17299","citing_title":"Toward Principled LLM Safety Testing: Solving the Jailbreak Oracle Problem","ref_index":37,"is_internal_anchor":false},{"citing_arxiv_id":"2401.05561","citing_title":"TrustLLM: Trustworthiness in Large Language Models","ref_index":245,"is_internal_anchor":false},{"citing_arxiv_id":"2310.02446","citing_title":"Low-Resource Languages Jailbreak GPT-4","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2402.10260","citing_title":"A StrongREJECT for Empty Jailbreaks","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2602.11528","citing_title":"Stop Tracking Me! Proactive Defense Against Attribute Inference Attack in LLMs","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2404.01318","citing_title":"JailbreakBench: An Open Robustness Benchmark for Jailbreaking Large Language Models","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2407.04295","citing_title":"Jailbreak Attacks and Defenses Against Large Language Models: A Survey","ref_index":46,"is_internal_anchor":false},{"citing_arxiv_id":"2604.23130","citing_title":"From Concept-Aligned Tokens to Vulnerable Features: Mechanistic Localization of Jailbreaks","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2502.18864","citing_title":"Towards an AI co-scientist","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2604.11309","citing_title":"The Salami Slicing Threat: Exploiting Cumulative Risks in LLM Systems","ref_index":16,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/OTA2I4NJZCJ5VTPATZ4RIQKDKQ","json":"https://pith.science/pith/OTA2I4NJZCJ5VTPATZ4RIQKDKQ.json","graph_json":"https://pith.science/api/pith-number/OTA2I4NJZCJ5VTPATZ4RIQKDKQ/graph.json","events_json":"https://pith.science/api/pith-number/OTA2I4NJZCJ5VTPATZ4RIQKDKQ/events.json","paper":"https://pith.science/paper/OTA2I4NJ"},"agent_actions":{"view_html":"https://pith.science/pith/OTA2I4NJZCJ5VTPATZ4RIQKDKQ","download_json":"https://pith.science/pith/OTA2I4NJZCJ5VTPATZ4RIQKDKQ.json","view_paper":"https://pith.science/paper/OTA2I4NJ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2309.01446&json=true","fetch_graph":"https://pith.science/api/pith-number/OTA2I4NJZCJ5VTPATZ4RIQKDKQ/graph.json","fetch_events":"https://pith.science/api/pith-number/OTA2I4NJZCJ5VTPATZ4RIQKDKQ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/OTA2I4NJZCJ5VTPATZ4RIQKDKQ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/OTA2I4NJZCJ5VTPATZ4RIQKDKQ/action/storage_attestation","attest_author":"https://pith.science/pith/OTA2I4NJZCJ5VTPATZ4RIQKDKQ/action/author_attestation","sign_citation":"https://pith.science/pith/OTA2I4NJZCJ5VTPATZ4RIQKDKQ/action/citation_signature","submit_replication":"https://pith.science/pith/OTA2I4NJZCJ5VTPATZ4RIQKDKQ/action/replication_record"}},"created_at":"2026-07-05T08:51:43.352896+00:00","updated_at":"2026-07-05T08:51:43.352896+00:00"}