{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:XWSBVZBK5G524AZ3T63S6EYV4U","short_pith_number":"pith:XWSBVZBK","schema_version":"1.0","canonical_sha256":"bda41ae42ae9bbae033b9fb72f1315e518ae8f3e5dd452a8570d546dc76d60d5","source":{"kind":"arxiv","id":"2311.07689","version":1},"attestation_state":"computed","paper":{"title":"MART: Improving LLM Safety with Multi-round Automatic Red-Teaming","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Chunting Zhou, Jiawei Han, Madian Khabsa, Qifan Wang, Rui Hou, Suyu Ge, Yi-Chia Wang, Yuning Mao","submitted_at":"2023-11-13T19:13:29Z","abstract_excerpt":"Red-teaming is a common practice for mitigating unsafe behaviors in Large Language Models (LLMs), which involves thoroughly assessing LLMs to identify potential flaws and addressing them with responsible and accurate responses. While effective, manual red-teaming is costly, and existing automatic red-teaming typically discovers safety risks without addressing them. In this paper, we propose a Multi-round Automatic Red-Teaming (MART) method, which incorporates both automatic adversarial prompt writing and safe response generation, significantly increasing red-teaming scalability and the safety "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2311.07689","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2023-11-13T19:13:29Z","cross_cats_sorted":[],"title_canon_sha256":"7ebc4247e6fd0940dbce7b0b3258bf3f805eb7e88b4ea1a2995a8757a0d75eb9","abstract_canon_sha256":"51584e8daa17c65adf7686551895ba921b7dcdb973523ed15de25167ce971953"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:12:22.420891Z","signature_b64":"VS/NVFh2DADeShIgGq9LJ85iEcocbY/UzK+KHAQqrg1buWRuPceIQ1anWeUcJCjpKlYm/knDWyG4G23GMKn6CQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"bda41ae42ae9bbae033b9fb72f1315e518ae8f3e5dd452a8570d546dc76d60d5","last_reissued_at":"2026-07-05T07:12:22.420448Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:12:22.420448Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"MART: Improving LLM Safety with Multi-round Automatic Red-Teaming","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Chunting Zhou, Jiawei Han, Madian Khabsa, Qifan Wang, Rui Hou, Suyu Ge, Yi-Chia Wang, Yuning Mao","submitted_at":"2023-11-13T19:13:29Z","abstract_excerpt":"Red-teaming is a common practice for mitigating unsafe behaviors in Large Language Models (LLMs), which involves thoroughly assessing LLMs to identify potential flaws and addressing them with responsible and accurate responses. While effective, manual red-teaming is costly, and existing automatic red-teaming typically discovers safety risks without addressing them. In this paper, we propose a Multi-round Automatic Red-Teaming (MART) method, which incorporates both automatic adversarial prompt writing and safe response generation, significantly increasing red-teaming scalability and the safety "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2311.07689","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2311.07689/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2311.07689","created_at":"2026-07-05T07:12:22.420505+00:00"},{"alias_kind":"arxiv_version","alias_value":"2311.07689v1","created_at":"2026-07-05T07:12:22.420505+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2311.07689","created_at":"2026-07-05T07:12:22.420505+00:00"},{"alias_kind":"pith_short_12","alias_value":"XWSBVZBK5G52","created_at":"2026-07-05T07:12:22.420505+00:00"},{"alias_kind":"pith_short_16","alias_value":"XWSBVZBK5G524AZ3","created_at":"2026-07-05T07:12:22.420505+00:00"},{"alias_kind":"pith_short_8","alias_value":"XWSBVZBK","created_at":"2026-07-05T07:12:22.420505+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":9,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.25476","citing_title":"A Red Teaming Framework for Large Language Models: A Case Study on Faithfulness Evaluation","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2606.27147","citing_title":"Safe Autoregressive Image Generation with Iterative Self-Improving Codebooks","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2408.12935","citing_title":"AI Safety Landscape for Large Language Models: Taxonomy, State-of-the-art, and Future Directions","ref_index":237,"is_internal_anchor":false},{"citing_arxiv_id":"2510.09689","citing_title":"When Search Goes Wrong: Red-Teaming Web-Augmented Large Language Models","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2407.04295","citing_title":"Jailbreak Attacks and Defenses Against Large Language Models: A Survey","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2604.21860","citing_title":"Transient Turn Injection: Exposing Stateless Multi-Turn Vulnerabilities in Large Language Models","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2604.21159","citing_title":"Adaptive Instruction Composition for Automated LLM Red-Teaming","ref_index":46,"is_internal_anchor":false},{"citing_arxiv_id":"2604.18946","citing_title":"Reasoning Structure Matters for Safety Alignment of Reasoning Models","ref_index":50,"is_internal_anchor":false},{"citing_arxiv_id":"2605.03179","citing_title":"A Validated Prompt Bank for Malicious Code Generation: Separating Executable Weapons from Security Knowledge in 1,554 Consensus-Labeled Prompts","ref_index":39,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/XWSBVZBK5G524AZ3T63S6EYV4U","json":"https://pith.science/pith/XWSBVZBK5G524AZ3T63S6EYV4U.json","graph_json":"https://pith.science/api/pith-number/XWSBVZBK5G524AZ3T63S6EYV4U/graph.json","events_json":"https://pith.science/api/pith-number/XWSBVZBK5G524AZ3T63S6EYV4U/events.json","paper":"https://pith.science/paper/XWSBVZBK"},"agent_actions":{"view_html":"https://pith.science/pith/XWSBVZBK5G524AZ3T63S6EYV4U","download_json":"https://pith.science/pith/XWSBVZBK5G524AZ3T63S6EYV4U.json","view_paper":"https://pith.science/paper/XWSBVZBK","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2311.07689&json=true","fetch_graph":"https://pith.science/api/pith-number/XWSBVZBK5G524AZ3T63S6EYV4U/graph.json","fetch_events":"https://pith.science/api/pith-number/XWSBVZBK5G524AZ3T63S6EYV4U/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/XWSBVZBK5G524AZ3T63S6EYV4U/action/timestamp_anchor","attest_storage":"https://pith.science/pith/XWSBVZBK5G524AZ3T63S6EYV4U/action/storage_attestation","attest_author":"https://pith.science/pith/XWSBVZBK5G524AZ3T63S6EYV4U/action/author_attestation","sign_citation":"https://pith.science/pith/XWSBVZBK5G524AZ3T63S6EYV4U/action/citation_signature","submit_replication":"https://pith.science/pith/XWSBVZBK5G524AZ3T63S6EYV4U/action/replication_record"}},"created_at":"2026-07-05T07:12:22.420505+00:00","updated_at":"2026-07-05T07:12:22.420505+00:00"}