{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:7QJSIVYC7MGR273PLZFKS4RTWX","short_pith_number":"pith:7QJSIVYC","schema_version":"1.0","canonical_sha256":"fc13245702fb0d1d7f6f5e4aa97233b5efa366f793d0dc7b84f0c4bb7dc70241","source":{"kind":"arxiv","id":"2408.03603","version":1},"attestation_state":"computed","paper":{"title":"EnJa: Ensemble Jailbreak on Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.LG"],"primary_cat":"cs.CR","authors_text":"Jiahao Zhang, Ruofan Wang, Xingjun Ma, Yu-Gang Jiang, Zilong Wang","submitted_at":"2024-08-07T07:46:08Z","abstract_excerpt":"As Large Language Models (LLMs) are increasingly being deployed in safety-critical applications, their vulnerability to potential jailbreaks -- malicious prompts that can disable the safety mechanism of LLMs -- has attracted growing research attention. While alignment methods have been proposed to protect LLMs from jailbreaks, many have found that aligned LLMs can still be jailbroken by carefully crafted malicious prompts, producing content that violates policy regulations. Existing jailbreak attacks on LLMs can be categorized into prompt-level methods which make up stories/logic to circumvent"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2408.03603","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CR","submitted_at":"2024-08-07T07:46:08Z","cross_cats_sorted":["cs.AI","cs.CL","cs.LG"],"title_canon_sha256":"3c406f274224d5c5f805b6913a963333b60a51acf2e907a243f4ec3e85f1f2cb","abstract_canon_sha256":"2f5736bd8fa388f4bc6013e88b3721546ec7cc746e90bd749220593811c80163"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:53:00.777201Z","signature_b64":"tri62KSuD9CLGxrMSaeSv3W8/sfzYRKSl20HoHX74Q4wsb75u7ASTPI0g2O8T3Gwd5wWQpxonQe/u5UJo/IWAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"fc13245702fb0d1d7f6f5e4aa97233b5efa366f793d0dc7b84f0c4bb7dc70241","last_reissued_at":"2026-07-05T08:53:00.776750Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:53:00.776750Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"EnJa: Ensemble Jailbreak on Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.LG"],"primary_cat":"cs.CR","authors_text":"Jiahao Zhang, Ruofan Wang, Xingjun Ma, Yu-Gang Jiang, Zilong Wang","submitted_at":"2024-08-07T07:46:08Z","abstract_excerpt":"As Large Language Models (LLMs) are increasingly being deployed in safety-critical applications, their vulnerability to potential jailbreaks -- malicious prompts that can disable the safety mechanism of LLMs -- has attracted growing research attention. While alignment methods have been proposed to protect LLMs from jailbreaks, many have found that aligned LLMs can still be jailbroken by carefully crafted malicious prompts, producing content that violates policy regulations. Existing jailbreak attacks on LLMs can be categorized into prompt-level methods which make up stories/logic to circumvent"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2408.03603","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2408.03603/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2408.03603","created_at":"2026-07-05T08:53:00.776807+00:00"},{"alias_kind":"arxiv_version","alias_value":"2408.03603v1","created_at":"2026-07-05T08:53:00.776807+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2408.03603","created_at":"2026-07-05T08:53:00.776807+00:00"},{"alias_kind":"pith_short_12","alias_value":"7QJSIVYC7MGR","created_at":"2026-07-05T08:53:00.776807+00:00"},{"alias_kind":"pith_short_16","alias_value":"7QJSIVYC7MGR273P","created_at":"2026-07-05T08:53:00.776807+00:00"},{"alias_kind":"pith_short_8","alias_value":"7QJSIVYC","created_at":"2026-07-05T08:53:00.776807+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2502.05206","citing_title":"Safety at Scale: A Comprehensive Survey of Large Model and Agent Safety","ref_index":89,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/7QJSIVYC7MGR273PLZFKS4RTWX","json":"https://pith.science/pith/7QJSIVYC7MGR273PLZFKS4RTWX.json","graph_json":"https://pith.science/api/pith-number/7QJSIVYC7MGR273PLZFKS4RTWX/graph.json","events_json":"https://pith.science/api/pith-number/7QJSIVYC7MGR273PLZFKS4RTWX/events.json","paper":"https://pith.science/paper/7QJSIVYC"},"agent_actions":{"view_html":"https://pith.science/pith/7QJSIVYC7MGR273PLZFKS4RTWX","download_json":"https://pith.science/pith/7QJSIVYC7MGR273PLZFKS4RTWX.json","view_paper":"https://pith.science/paper/7QJSIVYC","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2408.03603&json=true","fetch_graph":"https://pith.science/api/pith-number/7QJSIVYC7MGR273PLZFKS4RTWX/graph.json","fetch_events":"https://pith.science/api/pith-number/7QJSIVYC7MGR273PLZFKS4RTWX/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/7QJSIVYC7MGR273PLZFKS4RTWX/action/timestamp_anchor","attest_storage":"https://pith.science/pith/7QJSIVYC7MGR273PLZFKS4RTWX/action/storage_attestation","attest_author":"https://pith.science/pith/7QJSIVYC7MGR273PLZFKS4RTWX/action/author_attestation","sign_citation":"https://pith.science/pith/7QJSIVYC7MGR273PLZFKS4RTWX/action/citation_signature","submit_replication":"https://pith.science/pith/7QJSIVYC7MGR273PLZFKS4RTWX/action/replication_record"}},"created_at":"2026-07-05T08:53:00.776807+00:00","updated_at":"2026-07-05T08:53:00.776807+00:00"}