{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:U6AFBKJ3BX4PO373ITXYPGRZHZ","short_pith_number":"pith:U6AFBKJ3","schema_version":"1.0","canonical_sha256":"a78050a93b0df8f76ffb44ef879a393e5dbc5e90613652af2ed4b10c5b677336","source":{"kind":"arxiv","id":"2307.08715","version":2},"attestation_state":"computed","paper":{"title":"MasterKey: Automated Jailbreak Across Multiple Large Language Model Chatbots","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CR","authors_text":"Gelei Deng, Haoyu Wang, Kailong Wang, Tianwei Zhang, Yang Liu, Yi Liu, Ying Zhang, Yuekang Li, Zefeng Li","submitted_at":"2023-07-16T01:07:15Z","abstract_excerpt":"Large Language Models (LLMs) have revolutionized Artificial Intelligence (AI) services due to their exceptional proficiency in understanding and generating human-like text. LLM chatbots, in particular, have seen widespread adoption, transforming human-machine interactions. However, these LLM chatbots are susceptible to \"jailbreak\" attacks, where malicious users manipulate prompts to elicit inappropriate or sensitive responses, contravening service policies. Despite existing attempts to mitigate such threats, our research reveals a substantial gap in our understanding of these vulnerabilities, "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2307.08715","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CR","submitted_at":"2023-07-16T01:07:15Z","cross_cats_sorted":[],"title_canon_sha256":"fb07a1049b7126cad0a86f2c71b0092f8c50e41f75d954f6fa633c3c932d2056","abstract_canon_sha256":"2406aec8d05e402eaf701a347de84df144f8e9a15ac1eb66f044c968347acebe"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:44:19.135302Z","signature_b64":"OB7iP69UCLHHIwQ57D3jPpoBD1xd0o1upKiL/bH1pynyZ0eJzlhvsd9WzAajiuJ8Lt6ZtU1uiI+IvCo+C0aABA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"a78050a93b0df8f76ffb44ef879a393e5dbc5e90613652af2ed4b10c5b677336","last_reissued_at":"2026-07-05T07:44:19.134884Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:44:19.134884Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"MasterKey: Automated Jailbreak Across Multiple Large Language Model Chatbots","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CR","authors_text":"Gelei Deng, Haoyu Wang, Kailong Wang, Tianwei Zhang, Yang Liu, Yi Liu, Ying Zhang, Yuekang Li, Zefeng Li","submitted_at":"2023-07-16T01:07:15Z","abstract_excerpt":"Large Language Models (LLMs) have revolutionized Artificial Intelligence (AI) services due to their exceptional proficiency in understanding and generating human-like text. LLM chatbots, in particular, have seen widespread adoption, transforming human-machine interactions. However, these LLM chatbots are susceptible to \"jailbreak\" attacks, where malicious users manipulate prompts to elicit inappropriate or sensitive responses, contravening service policies. Despite existing attempts to mitigate such threats, our research reveals a substantial gap in our understanding of these vulnerabilities, "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2307.08715","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2307.08715/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2307.08715","created_at":"2026-07-05T07:44:19.134943+00:00"},{"alias_kind":"arxiv_version","alias_value":"2307.08715v2","created_at":"2026-07-05T07:44:19.134943+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2307.08715","created_at":"2026-07-05T07:44:19.134943+00:00"},{"alias_kind":"pith_short_12","alias_value":"U6AFBKJ3BX4P","created_at":"2026-07-05T07:44:19.134943+00:00"},{"alias_kind":"pith_short_16","alias_value":"U6AFBKJ3BX4PO373","created_at":"2026-07-05T07:44:19.134943+00:00"},{"alias_kind":"pith_short_8","alias_value":"U6AFBKJ3","created_at":"2026-07-05T07:44:19.134943+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":21,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.09700","citing_title":"What the Eyes See, the LLMs Miss: Exploiting Human Perception for Adversarial Text Attacks","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2606.05609","citing_title":"SlotGCG: Exploiting the Positional Vulnerability in LLMs for Jailbreak Attacks","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2402.06922","citing_title":"Whispers in the Machine: Confidentiality in Agentic Systems","ref_index":51,"is_internal_anchor":false},{"citing_arxiv_id":"2603.04459","citing_title":"Benchmark of Benchmarks: Unpacking Influence and Code Repository Quality in LLM Safety Benchmarks","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2605.21362","citing_title":"LASH: Adaptive Semantic Hybridization for Black-Box Jailbreaking of Large Language Models","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2508.20325","citing_title":"GUARD: Guideline Upholding Test through Adaptive Role-play and Jailbreak Diagnostics for LLMs","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2510.16558","citing_title":"A First Look at the Security Issues in the Model Context Protocol Ecosystem","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2404.01833","citing_title":"Great, Now Write an Article About That: The Crescendo Multi-Turn LLM Jailbreak Attack","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2308.03825","citing_title":"\"Do Anything Now\": Characterizing and Evaluating In-The-Wild Jailbreak Prompts on Large Language Models","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2402.10260","citing_title":"A StrongREJECT for Empty Jailbreaks","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2602.02280","citing_title":"RACC: Representation-Aware Coverage Criteria for LLM Safety Testing","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2309.10253","citing_title":"GPTFUZZER: Red Teaming Large Language Models with Auto-Generated Jailbreak Prompts","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2407.04295","citing_title":"Jailbreak Attacks and Defenses Against Large Language Models: A Survey","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2604.04060","citing_title":"CoopGuard: Stateful Cooperative Agents Safeguarding LLMs Against Evolving Multi-Round Attacks","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05868","citing_title":"SkillScope: Toward Fine-Grained Least-Privilege Enforcement for Agent Skills","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2604.10326","citing_title":"Jailbreaking the Matrix: Nullspace Steering for Controlled Model Subversion","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2604.12168","citing_title":"Fully Homomorphic Encryption on Llama 3 model for privacy preserving LLM inference","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2604.08846","citing_title":"Dictionary-Aligned Concept Control for Safeguarding Multimodal LLMs","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2604.05498","citing_title":"JailWAM: Jailbreaking World Action Models in Robot Control","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2604.15780","citing_title":"Pruning Unsafe Tickets: A Resource-Efficient Framework for Safer and More Robust LLMs","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2605.02187","citing_title":"Rewriting the Response Path: Silent Tampering and Provider-Signed Defense in BYOK LLM Agents","ref_index":55,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/U6AFBKJ3BX4PO373ITXYPGRZHZ","json":"https://pith.science/pith/U6AFBKJ3BX4PO373ITXYPGRZHZ.json","graph_json":"https://pith.science/api/pith-number/U6AFBKJ3BX4PO373ITXYPGRZHZ/graph.json","events_json":"https://pith.science/api/pith-number/U6AFBKJ3BX4PO373ITXYPGRZHZ/events.json","paper":"https://pith.science/paper/U6AFBKJ3"},"agent_actions":{"view_html":"https://pith.science/pith/U6AFBKJ3BX4PO373ITXYPGRZHZ","download_json":"https://pith.science/pith/U6AFBKJ3BX4PO373ITXYPGRZHZ.json","view_paper":"https://pith.science/paper/U6AFBKJ3","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2307.08715&json=true","fetch_graph":"https://pith.science/api/pith-number/U6AFBKJ3BX4PO373ITXYPGRZHZ/graph.json","fetch_events":"https://pith.science/api/pith-number/U6AFBKJ3BX4PO373ITXYPGRZHZ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/U6AFBKJ3BX4PO373ITXYPGRZHZ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/U6AFBKJ3BX4PO373ITXYPGRZHZ/action/storage_attestation","attest_author":"https://pith.science/pith/U6AFBKJ3BX4PO373ITXYPGRZHZ/action/author_attestation","sign_citation":"https://pith.science/pith/U6AFBKJ3BX4PO373ITXYPGRZHZ/action/citation_signature","submit_replication":"https://pith.science/pith/U6AFBKJ3BX4PO373ITXYPGRZHZ/action/replication_record"}},"created_at":"2026-07-05T07:44:19.134943+00:00","updated_at":"2026-07-05T07:44:19.134943+00:00"}