{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:QW5O4CD2DVGJ4XEI5K7YWSJY7Z","short_pith_number":"pith:QW5O4CD2","schema_version":"1.0","canonical_sha256":"85baee087a1d4c9e5c88eabf8b4938fe61eb1f41ac8a902d8575b611db2a3a62","source":{"kind":"arxiv","id":"2412.18693","version":1},"attestation_state":"computed","paper":{"title":"Diverse and Effective Red Teaming with Auto-generated Rewards and Multi-step Reinforcement Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Alex Beutel, Johannes Heidecke, Kai Xiao, Lilian Weng","submitted_at":"2024-12-24T22:38:46Z","abstract_excerpt":"Automated red teaming can discover rare model failures and generate challenging examples that can be used for training or evaluation. However, a core challenge in automated red teaming is ensuring that the attacks are both diverse and effective. Prior methods typically succeed in optimizing either for diversity or for effectiveness, but rarely both. In this paper, we provide methods that enable automated red teaming to generate a large number of diverse and successful attacks.\n  Our approach decomposes the task into two steps: (1) automated methods for generating diverse attack goals and (2) g"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2412.18693","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-12-24T22:38:46Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"ea06217b134cd06f0f3804cdbc8385866bb719811b57d6f700dadf91f79b011f","abstract_canon_sha256":"3c5182987b8e36727f4d690c8b486056a63358d4d9e90ea57ccff41d6a914144"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:54:09.193469Z","signature_b64":"esgsHLN1GbQbdMlYC/uBbAZ+ZcRYiZaC+soGerc3CB2FXF6koTX/yu0bPkZsx4TrPF8SQypvgGIIN/h2XifDCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"85baee087a1d4c9e5c88eabf8b4938fe61eb1f41ac8a902d8575b611db2a3a62","last_reissued_at":"2026-07-05T09:54:09.193043Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:54:09.193043Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Diverse and Effective Red Teaming with Auto-generated Rewards and Multi-step Reinforcement Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Alex Beutel, Johannes Heidecke, Kai Xiao, Lilian Weng","submitted_at":"2024-12-24T22:38:46Z","abstract_excerpt":"Automated red teaming can discover rare model failures and generate challenging examples that can be used for training or evaluation. However, a core challenge in automated red teaming is ensuring that the attacks are both diverse and effective. Prior methods typically succeed in optimizing either for diversity or for effectiveness, but rarely both. In this paper, we provide methods that enable automated red teaming to generate a large number of diverse and successful attacks.\n  Our approach decomposes the task into two steps: (1) automated methods for generating diverse attack goals and (2) g"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2412.18693","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2412.18693/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2412.18693","created_at":"2026-07-05T09:54:09.193098+00:00"},{"alias_kind":"arxiv_version","alias_value":"2412.18693v1","created_at":"2026-07-05T09:54:09.193098+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2412.18693","created_at":"2026-07-05T09:54:09.193098+00:00"},{"alias_kind":"pith_short_12","alias_value":"QW5O4CD2DVGJ","created_at":"2026-07-05T09:54:09.193098+00:00"},{"alias_kind":"pith_short_16","alias_value":"QW5O4CD2DVGJ4XEI","created_at":"2026-07-05T09:54:09.193098+00:00"},{"alias_kind":"pith_short_8","alias_value":"QW5O4CD2","created_at":"2026-07-05T09:54:09.193098+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2510.07239","citing_title":"Red-Bandit: Test-Time Adaptation for LLM Red-Teaming via Bandit-Guided LoRA Experts","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2507.06261","citing_title":"Gemini 2.5: Pushing the Frontier with Advanced Reasoning, Multimodality, Long Context, and Next Generation Agentic Capabilities","ref_index":8,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/QW5O4CD2DVGJ4XEI5K7YWSJY7Z","json":"https://pith.science/pith/QW5O4CD2DVGJ4XEI5K7YWSJY7Z.json","graph_json":"https://pith.science/api/pith-number/QW5O4CD2DVGJ4XEI5K7YWSJY7Z/graph.json","events_json":"https://pith.science/api/pith-number/QW5O4CD2DVGJ4XEI5K7YWSJY7Z/events.json","paper":"https://pith.science/paper/QW5O4CD2"},"agent_actions":{"view_html":"https://pith.science/pith/QW5O4CD2DVGJ4XEI5K7YWSJY7Z","download_json":"https://pith.science/pith/QW5O4CD2DVGJ4XEI5K7YWSJY7Z.json","view_paper":"https://pith.science/paper/QW5O4CD2","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2412.18693&json=true","fetch_graph":"https://pith.science/api/pith-number/QW5O4CD2DVGJ4XEI5K7YWSJY7Z/graph.json","fetch_events":"https://pith.science/api/pith-number/QW5O4CD2DVGJ4XEI5K7YWSJY7Z/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/QW5O4CD2DVGJ4XEI5K7YWSJY7Z/action/timestamp_anchor","attest_storage":"https://pith.science/pith/QW5O4CD2DVGJ4XEI5K7YWSJY7Z/action/storage_attestation","attest_author":"https://pith.science/pith/QW5O4CD2DVGJ4XEI5K7YWSJY7Z/action/author_attestation","sign_citation":"https://pith.science/pith/QW5O4CD2DVGJ4XEI5K7YWSJY7Z/action/citation_signature","submit_replication":"https://pith.science/pith/QW5O4CD2DVGJ4XEI5K7YWSJY7Z/action/replication_record"}},"created_at":"2026-07-05T09:54:09.193098+00:00","updated_at":"2026-07-05T09:54:09.193098+00:00"}