{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:ZIKDHTKR7G2J6THFEU555EWSPF","short_pith_number":"pith:ZIKDHTKR","schema_version":"1.0","canonical_sha256":"ca1433cd51f9b49f4ce5253bde92d2794e8237b1a9d2e61115bd4d95bdd5dd72","source":{"kind":"arxiv","id":"2310.00322","version":5},"attestation_state":"computed","paper":{"title":"Evolving Diverse Red-team Language Models in Multi-round Multi-agent Games","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.GT"],"primary_cat":"cs.CL","authors_text":"Chengdong Ma, Hai Ci, Jun Gao, Minquan Gao, Xuehai Pan, Yaodong Yang, Ziran Yang","submitted_at":"2023-09-30T09:35:50Z","abstract_excerpt":"The primary challenge in deploying Large Language Model (LLM) is ensuring its harmlessness. Red team can identify vulnerabilities by attacking LLM to attain safety. However, current efforts heavily rely on single-round prompt designs and unilateral red team optimizations against fixed blue teams. These static approaches lead to significant reductions in generation diversity, known as the mode collapse, which makes it difficult to discover the potential risks in the increasingly complex human-LLM interactions. Here we introduce dynamic Red Team Game (RTG) to comprehensively analyze the multi-ro"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2310.00322","kind":"arxiv","version":5},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2023-09-30T09:35:50Z","cross_cats_sorted":["cs.GT"],"title_canon_sha256":"4109f3f1d733580f9c2962f1cdfbb919525532f1f04ee810e38df863daba797b","abstract_canon_sha256":"227c277ffef6a9b815262b8adc823f856ca03626832f2e696454aa378d95d228"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:49:26.001012Z","signature_b64":"ctSs7S/ZG3cluTIe0qTVvFZmgQ9CO7PRKfJxe3kdc/pUbg+bwQdlWDVKei1u7Hi4cdBYoBnpl3vvnQJnKqt4BQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ca1433cd51f9b49f4ce5253bde92d2794e8237b1a9d2e61115bd4d95bdd5dd72","last_reissued_at":"2026-07-05T08:49:26.000547Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:49:26.000547Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Evolving Diverse Red-team Language Models in Multi-round Multi-agent Games","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.GT"],"primary_cat":"cs.CL","authors_text":"Chengdong Ma, Hai Ci, Jun Gao, Minquan Gao, Xuehai Pan, Yaodong Yang, Ziran Yang","submitted_at":"2023-09-30T09:35:50Z","abstract_excerpt":"The primary challenge in deploying Large Language Model (LLM) is ensuring its harmlessness. Red team can identify vulnerabilities by attacking LLM to attain safety. However, current efforts heavily rely on single-round prompt designs and unilateral red team optimizations against fixed blue teams. These static approaches lead to significant reductions in generation diversity, known as the mode collapse, which makes it difficult to discover the potential risks in the increasingly complex human-LLM interactions. Here we introduce dynamic Red Team Game (RTG) to comprehensively analyze the multi-ro"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2310.00322","kind":"arxiv","version":5},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2310.00322/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2310.00322","created_at":"2026-07-05T08:49:26.000607+00:00"},{"alias_kind":"arxiv_version","alias_value":"2310.00322v5","created_at":"2026-07-05T08:49:26.000607+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2310.00322","created_at":"2026-07-05T08:49:26.000607+00:00"},{"alias_kind":"pith_short_12","alias_value":"ZIKDHTKR7G2J","created_at":"2026-07-05T08:49:26.000607+00:00"},{"alias_kind":"pith_short_16","alias_value":"ZIKDHTKR7G2J6THF","created_at":"2026-07-05T08:49:26.000607+00:00"},{"alias_kind":"pith_short_8","alias_value":"ZIKDHTKR","created_at":"2026-07-05T08:49:26.000607+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2503.21460","citing_title":"Large Language Model Agent: A Survey on Methodology, Applications and Challenges","ref_index":98,"is_internal_anchor":false},{"citing_arxiv_id":"2605.19377","citing_title":"The Evaluation Game: Beyond Static LLM Benchmarking","ref_index":36,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ZIKDHTKR7G2J6THFEU555EWSPF","json":"https://pith.science/pith/ZIKDHTKR7G2J6THFEU555EWSPF.json","graph_json":"https://pith.science/api/pith-number/ZIKDHTKR7G2J6THFEU555EWSPF/graph.json","events_json":"https://pith.science/api/pith-number/ZIKDHTKR7G2J6THFEU555EWSPF/events.json","paper":"https://pith.science/paper/ZIKDHTKR"},"agent_actions":{"view_html":"https://pith.science/pith/ZIKDHTKR7G2J6THFEU555EWSPF","download_json":"https://pith.science/pith/ZIKDHTKR7G2J6THFEU555EWSPF.json","view_paper":"https://pith.science/paper/ZIKDHTKR","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2310.00322&json=true","fetch_graph":"https://pith.science/api/pith-number/ZIKDHTKR7G2J6THFEU555EWSPF/graph.json","fetch_events":"https://pith.science/api/pith-number/ZIKDHTKR7G2J6THFEU555EWSPF/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ZIKDHTKR7G2J6THFEU555EWSPF/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ZIKDHTKR7G2J6THFEU555EWSPF/action/storage_attestation","attest_author":"https://pith.science/pith/ZIKDHTKR7G2J6THFEU555EWSPF/action/author_attestation","sign_citation":"https://pith.science/pith/ZIKDHTKR7G2J6THFEU555EWSPF/action/citation_signature","submit_replication":"https://pith.science/pith/ZIKDHTKR7G2J6THFEU555EWSPF/action/replication_record"}},"created_at":"2026-07-05T08:49:26.000607+00:00","updated_at":"2026-07-05T08:49:26.000607+00:00"}