{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:BUQUHOYQPOI6KUHDSFFNZW5RMF","short_pith_number":"pith:BUQUHOYQ","schema_version":"1.0","canonical_sha256":"0d2143bb107b91e550e3914adcdbb16162b5ded1b051e53ea6cc35bc47c4594c","source":{"kind":"arxiv","id":"2310.15140","version":2},"attestation_state":"computed","paper":{"title":"AutoDAN: Interpretable Gradient-Based Adversarial Attacks on Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.LG"],"primary_cat":"cs.CR","authors_text":"Ani Nenkova, Bang An, Furong Huang, Gang Wu, Joe Barrow, Ruiyi Zhang, Sicheng Zhu, Tong Sun, Zichao Wang","submitted_at":"2023-10-23T17:46:07Z","abstract_excerpt":"Safety alignment of Large Language Models (LLMs) can be compromised with manual jailbreak attacks and (automatic) adversarial attacks. Recent studies suggest that defending against these attacks is possible: adversarial attacks generate unlimited but unreadable gibberish prompts, detectable by perplexity-based filters; manual jailbreak attacks craft readable prompts, but their limited number due to the necessity of human creativity allows for easy blocking. In this paper, we show that these solutions may be too optimistic. We introduce AutoDAN, an interpretable, gradient-based adversarial atta"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2310.15140","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CR","submitted_at":"2023-10-23T17:46:07Z","cross_cats_sorted":["cs.AI","cs.CL","cs.LG"],"title_canon_sha256":"d5b5989ed1e3ef5af4fd1d1e4b91d152f802aeb9b0ece8db9fefcc99fed9c59c","abstract_canon_sha256":"2cdd0660681f0f8b2a90adcb75086f7ad0d7c3e76cf3518c487d700a7cbee5ba"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:24:01.369099Z","signature_b64":"DkS9k/GcNPU2+VxxSNMezErfObUohFqYrRAB3CRM54LuEjFVnUDDnh8tiTSdAPRnoesCNa/YQZYI9QedyFLQCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"0d2143bb107b91e550e3914adcdbb16162b5ded1b051e53ea6cc35bc47c4594c","last_reissued_at":"2026-07-05T07:24:01.368566Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:24:01.368566Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"AutoDAN: Interpretable Gradient-Based Adversarial Attacks on Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.LG"],"primary_cat":"cs.CR","authors_text":"Ani Nenkova, Bang An, Furong Huang, Gang Wu, Joe Barrow, Ruiyi Zhang, Sicheng Zhu, Tong Sun, Zichao Wang","submitted_at":"2023-10-23T17:46:07Z","abstract_excerpt":"Safety alignment of Large Language Models (LLMs) can be compromised with manual jailbreak attacks and (automatic) adversarial attacks. Recent studies suggest that defending against these attacks is possible: adversarial attacks generate unlimited but unreadable gibberish prompts, detectable by perplexity-based filters; manual jailbreak attacks craft readable prompts, but their limited number due to the necessity of human creativity allows for easy blocking. In this paper, we show that these solutions may be too optimistic. We introduce AutoDAN, an interpretable, gradient-based adversarial atta"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2310.15140","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2310.15140/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2310.15140","created_at":"2026-07-05T07:24:01.368627+00:00"},{"alias_kind":"arxiv_version","alias_value":"2310.15140v2","created_at":"2026-07-05T07:24:01.368627+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2310.15140","created_at":"2026-07-05T07:24:01.368627+00:00"},{"alias_kind":"pith_short_12","alias_value":"BUQUHOYQPOI6","created_at":"2026-07-05T07:24:01.368627+00:00"},{"alias_kind":"pith_short_16","alias_value":"BUQUHOYQPOI6KUHD","created_at":"2026-07-05T07:24:01.368627+00:00"},{"alias_kind":"pith_short_8","alias_value":"BUQUHOYQ","created_at":"2026-07-05T07:24:01.368627+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":23,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.06326","citing_title":"DT-Guard: Intent-Driven Reasoning-Active Training for Reasoning-Free LLM Safety Guardrail","ref_index":6,"is_internal_anchor":true},{"citing_arxiv_id":"2606.25476","citing_title":"A Red Teaming Framework for Large Language Models: A Case Study on Faithfulness Evaluation","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2606.05523","citing_title":"CHASE: Adversarial Red-Blue Teaming for Improving LLM Safety using Reinforcement Learning","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2606.03647","citing_title":"Black-box, Adaptive, Efficient, Transferable, Harmful, Applicable... Attacks Are All You Need to Break LLMs","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08876","citing_title":"OTora: A Unified Red Teaming Framework for Reasoning-Level Denial-of-Service in LLM Agents","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2605.28664","citing_title":"Activation Steering for Synthetic Data Generation: The Role of Diversity in Downstream Safety Detection","ref_index":57,"is_internal_anchor":false},{"citing_arxiv_id":"2402.06922","citing_title":"Whispers in the Machine: Confidentiality in Agentic Systems","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2410.15362","citing_title":"Faster-GCG: Efficient Discrete Optimization Jailbreak Attacks against Aligned Large Language Models","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2504.20472","citing_title":"Robustness via Referencing: Defending against Prompt Injection Attacks by Referencing the Executed Instruction","ref_index":50,"is_internal_anchor":false},{"citing_arxiv_id":"2506.12382","citing_title":"Exploring the Secondary Risks of Large Language Models","ref_index":55,"is_internal_anchor":false},{"citing_arxiv_id":"2508.20325","citing_title":"GUARD: Guideline Upholding Test through Adaptive Role-play and Jailbreak Diagnostics for LLMs","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2510.09093","citing_title":"Exploiting Web Search Tools of AI Agents for Data Exfiltration","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2510.23883","citing_title":"Agentic AI Security: Threats, Defenses, Evaluation, and Open Challenges","ref_index":67,"is_internal_anchor":false},{"citing_arxiv_id":"2402.10260","citing_title":"A StrongREJECT for Empty Jailbreaks","ref_index":43,"is_internal_anchor":false},{"citing_arxiv_id":"2407.04295","citing_title":"Jailbreak Attacks and Defenses Against Large Language Models: A Survey","ref_index":124,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08876","citing_title":"OTora: A Unified Red Teaming Framework for Reasoning-Level Denial-of-Service in LLM Agents","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09225","citing_title":"The Art of the Jailbreak: Formulating Jailbreak Attacks for LLM Security Beyond Binary Scoring","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2604.18901","citing_title":"Harmful Intent as a Geometrically Recoverable Feature of LLM Residual Streams","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01899","citing_title":"Disentangling Intent from Role: Adversarial Self-Play for Persona-Invariant Safety Alignment","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2604.10326","citing_title":"Jailbreaking the Matrix: Nullspace Steering for Controlled Model Subversion","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2604.06811","citing_title":"SkillTrojan: Backdoor Attacks on Skill-Based Agent Systems","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2604.07727","citing_title":"TrajGuard: Streaming Hidden-state Trajectory Detection for Decoding-time Jailbreak Defense","ref_index":48,"is_internal_anchor":false},{"citing_arxiv_id":"2604.18901","citing_title":"Harmful Intent as a Geometrically Recoverable Feature of LLM Residual Streams","ref_index":14,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/BUQUHOYQPOI6KUHDSFFNZW5RMF","json":"https://pith.science/pith/BUQUHOYQPOI6KUHDSFFNZW5RMF.json","graph_json":"https://pith.science/api/pith-number/BUQUHOYQPOI6KUHDSFFNZW5RMF/graph.json","events_json":"https://pith.science/api/pith-number/BUQUHOYQPOI6KUHDSFFNZW5RMF/events.json","paper":"https://pith.science/paper/BUQUHOYQ"},"agent_actions":{"view_html":"https://pith.science/pith/BUQUHOYQPOI6KUHDSFFNZW5RMF","download_json":"https://pith.science/pith/BUQUHOYQPOI6KUHDSFFNZW5RMF.json","view_paper":"https://pith.science/paper/BUQUHOYQ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2310.15140&json=true","fetch_graph":"https://pith.science/api/pith-number/BUQUHOYQPOI6KUHDSFFNZW5RMF/graph.json","fetch_events":"https://pith.science/api/pith-number/BUQUHOYQPOI6KUHDSFFNZW5RMF/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/BUQUHOYQPOI6KUHDSFFNZW5RMF/action/timestamp_anchor","attest_storage":"https://pith.science/pith/BUQUHOYQPOI6KUHDSFFNZW5RMF/action/storage_attestation","attest_author":"https://pith.science/pith/BUQUHOYQPOI6KUHDSFFNZW5RMF/action/author_attestation","sign_citation":"https://pith.science/pith/BUQUHOYQPOI6KUHDSFFNZW5RMF/action/citation_signature","submit_replication":"https://pith.science/pith/BUQUHOYQPOI6KUHDSFFNZW5RMF/action/replication_record"}},"created_at":"2026-07-05T07:24:01.368627+00:00","updated_at":"2026-07-05T07:24:01.368627+00:00"}