{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:IE3I4HUQ33N6AXLYUHLTFD3LL5","short_pith_number":"pith:IE3I4HUQ","schema_version":"1.0","canonical_sha256":"41368e1e90dedbe05d78a1d7328f6b5f56a4f57af56711578e0a382efeb69c13","source":{"kind":"arxiv","id":"2403.04783","version":2},"attestation_state":"computed","paper":{"title":"AutoDefense: Multi-Agent LLM Defense against Jailbreak Attacks","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL","cs.CR"],"primary_cat":"cs.LG","authors_text":"Huazheng Wang, Qingyun Wu, Xiao Zhang, Yifan Zeng, Yiran Wu","submitted_at":"2024-03-02T16:52:22Z","abstract_excerpt":"Despite extensive pre-training in moral alignment to prevent generating harmful information, large language models (LLMs) remain vulnerable to jailbreak attacks. In this paper, we propose AutoDefense, a multi-agent defense framework that filters harmful responses from LLMs. With the response-filtering mechanism, our framework is robust against different jailbreak attack prompts, and can be used to defend different victim models. AutoDefense assigns different roles to LLM agents and employs them to complete the defense task collaboratively. The division in tasks enhances the overall instruction"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2403.04783","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-03-02T16:52:22Z","cross_cats_sorted":["cs.CL","cs.CR"],"title_canon_sha256":"da96d90fe66a607e18c59cb055307906bb59d01b9813c2990913344409cb3ee6","abstract_canon_sha256":"d7dfd4ab33ebc65dff5095319aea8ce52942134d567ec8e3cd635623a838c1bd"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:35:08.702751Z","signature_b64":"JHG4mvT0JqkLye41n1L6yDwin/SUWe8qYuKzXG7I2HCzDt0Dm/fAkcX4wyyArgvtaWuAfZ8l6rGPbEPL71osDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"41368e1e90dedbe05d78a1d7328f6b5f56a4f57af56711578e0a382efeb69c13","last_reissued_at":"2026-07-05T09:35:08.702300Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:35:08.702300Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"AutoDefense: Multi-Agent LLM Defense against Jailbreak Attacks","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL","cs.CR"],"primary_cat":"cs.LG","authors_text":"Huazheng Wang, Qingyun Wu, Xiao Zhang, Yifan Zeng, Yiran Wu","submitted_at":"2024-03-02T16:52:22Z","abstract_excerpt":"Despite extensive pre-training in moral alignment to prevent generating harmful information, large language models (LLMs) remain vulnerable to jailbreak attacks. In this paper, we propose AutoDefense, a multi-agent defense framework that filters harmful responses from LLMs. With the response-filtering mechanism, our framework is robust against different jailbreak attack prompts, and can be used to defend different victim models. AutoDefense assigns different roles to LLM agents and employs them to complete the defense task collaboratively. The division in tasks enhances the overall instruction"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2403.04783","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2403.04783/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2403.04783","created_at":"2026-07-05T09:35:08.702360+00:00"},{"alias_kind":"arxiv_version","alias_value":"2403.04783v2","created_at":"2026-07-05T09:35:08.702360+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2403.04783","created_at":"2026-07-05T09:35:08.702360+00:00"},{"alias_kind":"pith_short_12","alias_value":"IE3I4HUQ33N6","created_at":"2026-07-05T09:35:08.702360+00:00"},{"alias_kind":"pith_short_16","alias_value":"IE3I4HUQ33N6AXLY","created_at":"2026-07-05T09:35:08.702360+00:00"},{"alias_kind":"pith_short_8","alias_value":"IE3I4HUQ","created_at":"2026-07-05T09:35:08.702360+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":11,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.07461","citing_title":"Mitigating Taint-Style Vulnerabilities in MCP Servers via Security-Aware Tool Descriptions","ref_index":69,"is_internal_anchor":true},{"citing_arxiv_id":"2607.01277","citing_title":"Cognitive Firewall: A Proactive, Zero-Trust, Multi-Gate Framework for LLM Safety","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2605.26409","citing_title":"Jailbreak susceptibility prediction and mitigation via the behavioral geometry of models","ref_index":44,"is_internal_anchor":false},{"citing_arxiv_id":"2503.21460","citing_title":"Large Language Model Agent: A Survey on Methodology, Applications and Challenges","ref_index":188,"is_internal_anchor":false},{"citing_arxiv_id":"2506.17299","citing_title":"Toward Principled LLM Safety Testing: Solving the Jailbreak Oracle Problem","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2507.14201","citing_title":"ExCyTIn-Bench: Evaluating LLM agents on Cyber Threat Investigation","ref_index":59,"is_internal_anchor":false},{"citing_arxiv_id":"2510.22628","citing_title":"Sentra-Guard: A Real-Time Multilingual Defense Against Adversarial LLM Prompts","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2511.00181","citing_title":"From Evidence to Verdict: An Agent-Based Forensic Framework for AI-Generated Image Detection","ref_index":56,"is_internal_anchor":false},{"citing_arxiv_id":"2407.04295","citing_title":"Jailbreak Attacks and Defenses Against Large Language Models: A Survey","ref_index":110,"is_internal_anchor":false},{"citing_arxiv_id":"2604.24477","citing_title":"GAMMAF: A Common Framework for Graph-Based Anomaly Monitoring Benchmarking in LLM Multi-Agent Systems","ref_index":36,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05058","citing_title":"SoK: Robustness in Large Language Models against Jailbreak Attacks","ref_index":95,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/IE3I4HUQ33N6AXLYUHLTFD3LL5","json":"https://pith.science/pith/IE3I4HUQ33N6AXLYUHLTFD3LL5.json","graph_json":"https://pith.science/api/pith-number/IE3I4HUQ33N6AXLYUHLTFD3LL5/graph.json","events_json":"https://pith.science/api/pith-number/IE3I4HUQ33N6AXLYUHLTFD3LL5/events.json","paper":"https://pith.science/paper/IE3I4HUQ"},"agent_actions":{"view_html":"https://pith.science/pith/IE3I4HUQ33N6AXLYUHLTFD3LL5","download_json":"https://pith.science/pith/IE3I4HUQ33N6AXLYUHLTFD3LL5.json","view_paper":"https://pith.science/paper/IE3I4HUQ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2403.04783&json=true","fetch_graph":"https://pith.science/api/pith-number/IE3I4HUQ33N6AXLYUHLTFD3LL5/graph.json","fetch_events":"https://pith.science/api/pith-number/IE3I4HUQ33N6AXLYUHLTFD3LL5/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/IE3I4HUQ33N6AXLYUHLTFD3LL5/action/timestamp_anchor","attest_storage":"https://pith.science/pith/IE3I4HUQ33N6AXLYUHLTFD3LL5/action/storage_attestation","attest_author":"https://pith.science/pith/IE3I4HUQ33N6AXLYUHLTFD3LL5/action/author_attestation","sign_citation":"https://pith.science/pith/IE3I4HUQ33N6AXLYUHLTFD3LL5/action/citation_signature","submit_replication":"https://pith.science/pith/IE3I4HUQ33N6AXLYUHLTFD3LL5/action/replication_record"}},"created_at":"2026-07-05T09:35:08.702360+00:00","updated_at":"2026-07-05T09:35:08.702360+00:00"}