{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:ZKBWLQ5PZELTT5NR4SMQNO4HIL","short_pith_number":"pith:ZKBWLQ5P","schema_version":"1.0","canonical_sha256":"ca8365c3afc91739f5b1e49906bb8742e894fbb336e72ca6026791db0d4e0857","source":{"kind":"arxiv","id":"2503.15754","version":1},"attestation_state":"computed","paper":{"title":"AutoRedTeamer: Autonomous Red Teaming with Lifelong Attack Integration","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CR","authors_text":"Andy Zhou, Bo Li, Francesco Pinto, James Zou, Kevin Wu, Sanmi Koyejo, Shuang Yang, Yi Zeng, Yu Yang, Zhaorun Chen","submitted_at":"2025-03-20T00:13:04Z","abstract_excerpt":"As large language models (LLMs) become increasingly capable, security and safety evaluation are crucial. While current red teaming approaches have made strides in assessing LLM vulnerabilities, they often rely heavily on human input and lack comprehensive coverage of emerging attack vectors. This paper introduces AutoRedTeamer, a novel framework for fully automated, end-to-end red teaming against LLMs. AutoRedTeamer combines a multi-agent architecture with a memory-guided attack selection mechanism to enable continuous discovery and integration of new attack vectors. The dual-agent framework c"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2503.15754","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CR","submitted_at":"2025-03-20T00:13:04Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"fce990a933647cb8b275076c6710e7f264af00ab58210333b4cebed455d64740","abstract_canon_sha256":"c35942254f88db48b5b6b900c4dddcfc6a6b313c1b2839029478088a08de40dd"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:35:58.128547Z","signature_b64":"NGQiu43p9rEXXOdknLDE6GmMcBjVulhfhm53lTZX2WTQUOfFak7FM8v9nYeEqtggEHkYZN9j6DMv3CS1IyEhBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ca8365c3afc91739f5b1e49906bb8742e894fbb336e72ca6026791db0d4e0857","last_reissued_at":"2026-07-05T10:35:58.127877Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:35:58.127877Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"AutoRedTeamer: Autonomous Red Teaming with Lifelong Attack Integration","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CR","authors_text":"Andy Zhou, Bo Li, Francesco Pinto, James Zou, Kevin Wu, Sanmi Koyejo, Shuang Yang, Yi Zeng, Yu Yang, Zhaorun Chen","submitted_at":"2025-03-20T00:13:04Z","abstract_excerpt":"As large language models (LLMs) become increasingly capable, security and safety evaluation are crucial. While current red teaming approaches have made strides in assessing LLM vulnerabilities, they often rely heavily on human input and lack comprehensive coverage of emerging attack vectors. This paper introduces AutoRedTeamer, a novel framework for fully automated, end-to-end red teaming against LLMs. AutoRedTeamer combines a multi-agent architecture with a memory-guided attack selection mechanism to enable continuous discovery and integration of new attack vectors. The dual-agent framework c"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2503.15754","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2503.15754/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2503.15754","created_at":"2026-07-05T10:35:58.127956+00:00"},{"alias_kind":"arxiv_version","alias_value":"2503.15754v1","created_at":"2026-07-05T10:35:58.127956+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2503.15754","created_at":"2026-07-05T10:35:58.127956+00:00"},{"alias_kind":"pith_short_12","alias_value":"ZKBWLQ5PZELT","created_at":"2026-07-05T10:35:58.127956+00:00"},{"alias_kind":"pith_short_16","alias_value":"ZKBWLQ5PZELTT5NR","created_at":"2026-07-05T10:35:58.127956+00:00"},{"alias_kind":"pith_short_8","alias_value":"ZKBWLQ5P","created_at":"2026-07-05T10:35:58.127956+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":10,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.23892","citing_title":"REALM: A Unified Red-Teaming Benchmark for Physical-World VLMs","ref_index":45,"is_internal_anchor":false},{"citing_arxiv_id":"2606.02041","citing_title":"SentGuard: Sentence-Level Streaming Guardrails for Large Language Models","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2511.12710","citing_title":"Evolve the Method, Not the Prompts: Evolutionary Synthesis of Jailbreak Attacks on LLMs","ref_index":66,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17380","citing_title":"ADR: An Agentic Detection System for Enterprise Agentic AI Security","ref_index":39,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01970","citing_title":"Trojan Hippo: Weaponizing Agent Memory for Data Exfiltration","ref_index":94,"is_internal_anchor":false},{"citing_arxiv_id":"2509.26100","citing_title":"AgenticEval: Toward Agentic and Self-Evolving Safety Evaluation of Large Language Models","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11891","citing_title":"Proteus: A Self-Evolving Red Team for Agent Skill Ecosystems","ref_index":41,"is_internal_anchor":false},{"citing_arxiv_id":"2605.04019","citing_title":"Redefining AI Red Teaming in the Agentic Era: From Weeks to Hours","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2604.05549","citing_title":"Stop Fixating on Prompts: Reasoning Hijacking and Constraint Tightening for Red-Teaming LLM Agents","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2604.20704","citing_title":"Auto-ART: Structured Literature Synthesis and Automated Adversarial Robustness Testing","ref_index":32,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ZKBWLQ5PZELTT5NR4SMQNO4HIL","json":"https://pith.science/pith/ZKBWLQ5PZELTT5NR4SMQNO4HIL.json","graph_json":"https://pith.science/api/pith-number/ZKBWLQ5PZELTT5NR4SMQNO4HIL/graph.json","events_json":"https://pith.science/api/pith-number/ZKBWLQ5PZELTT5NR4SMQNO4HIL/events.json","paper":"https://pith.science/paper/ZKBWLQ5P"},"agent_actions":{"view_html":"https://pith.science/pith/ZKBWLQ5PZELTT5NR4SMQNO4HIL","download_json":"https://pith.science/pith/ZKBWLQ5PZELTT5NR4SMQNO4HIL.json","view_paper":"https://pith.science/paper/ZKBWLQ5P","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2503.15754&json=true","fetch_graph":"https://pith.science/api/pith-number/ZKBWLQ5PZELTT5NR4SMQNO4HIL/graph.json","fetch_events":"https://pith.science/api/pith-number/ZKBWLQ5PZELTT5NR4SMQNO4HIL/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ZKBWLQ5PZELTT5NR4SMQNO4HIL/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ZKBWLQ5PZELTT5NR4SMQNO4HIL/action/storage_attestation","attest_author":"https://pith.science/pith/ZKBWLQ5PZELTT5NR4SMQNO4HIL/action/author_attestation","sign_citation":"https://pith.science/pith/ZKBWLQ5PZELTT5NR4SMQNO4HIL/action/citation_signature","submit_replication":"https://pith.science/pith/ZKBWLQ5PZELTT5NR4SMQNO4HIL/action/replication_record"}},"created_at":"2026-07-05T10:35:58.127956+00:00","updated_at":"2026-07-05T10:35:58.127956+00:00"}