{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2022:QJDY4Y3ZXGOO52CUBNCGKNP6RV","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"1ee4e6d94955492a585c6de94a1450e146e1c56b12e6b5d11fda444ee1f08518","cross_cats_sorted":["cs.AI","cs.CR","cs.LG"],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2022-02-07T15:22:17Z","title_canon_sha256":"bb80b8b6cea33a49c34048dc37f0a2a6ceb18513c8d0fdde3c17a667d6323388"},"schema_version":"1.0","source":{"id":"2202.03286","kind":"arxiv","version":1}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2202.03286","created_at":"2026-07-05T03:54:37Z"},{"alias_kind":"arxiv_version","alias_value":"2202.03286v1","created_at":"2026-07-05T03:54:37Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2202.03286","created_at":"2026-07-05T03:54:37Z"},{"alias_kind":"pith_short_12","alias_value":"QJDY4Y3ZXGOO","created_at":"2026-07-05T03:54:37Z"},{"alias_kind":"pith_short_16","alias_value":"QJDY4Y3ZXGOO52CU","created_at":"2026-07-05T03:54:37Z"},{"alias_kind":"pith_short_8","alias_value":"QJDY4Y3Z","created_at":"2026-07-05T03:54:37Z"}],"graph_snapshots":[{"event_id":"sha256:0e538566dee65066a44547e32bbe18bb39326b2f070871251c9be6b5d988602d","target":"graph","created_at":"2026-07-05T03:54:37Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":4,"items":[{"attestation":"unclaimed","claim_id":"C1","kind":"strongest_claim","source":"verdict.strongest_claim","status":"machine_extracted","text":"we automatically find cases where a target LM behaves in a harmful way, by generating test cases (red teaming) using another LM... uncovering tens of thousands of offensive replies in a 280B parameter LM chatbot."},{"attestation":"unclaimed","claim_id":"C2","kind":"weakest_assumption","source":"verdict.weakest_assumption","status":"machine_extracted","text":"The classifier trained to detect offensive content accurately identifies the relevant harms, and the LM-generated test cases are sufficiently diverse, difficult, and representative of real user interactions."},{"attestation":"unclaimed","claim_id":"C3","kind":"one_line_summary","source":"verdict.one_line_summary","status":"machine_extracted","text":"One language model can generate diverse test cases to automatically uncover tens of thousands of harmful behaviors, including offensive replies and privacy leaks, in a large target language model."},{"attestation":"unclaimed","claim_id":"C4","kind":"headline","source":"verdict.pith_extraction.headline","status":"machine_extracted","text":"One language model generates test cases to automatically uncover tens of thousands of harmful behaviors in another language model."}],"snapshot_sha256":"1ca28634210af0459666ed92db628c2347a7c24b040f890a43e22d1a79d77650"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2202.03286/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Language Models (LMs) often cannot be deployed because of their potential to harm users in hard-to-predict ways. Prior work identifies harmful behaviors before deployment by using human annotators to hand-write test cases. However, human annotation is expensive, limiting the number and diversity of test cases. In this work, we automatically find cases where a target LM behaves in a harmful way, by generating test cases (\"red teaming\") using another LM. We evaluate the target LM's replies to generated test questions using a classifier trained to detect offensive content, uncovering tens of thou","authors_text":"Amelia Glaese, Ethan Perez, Francis Song, Geoffrey Irving, John Aslanides, Nat McAleese, Roman Ring, Saffron Huang, Trevor Cai","cross_cats":["cs.AI","cs.CR","cs.LG"],"headline":"One language model generates test cases to automatically uncover tens of thousands of harmful behaviors in another language model.","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2022-02-07T15:22:17Z","title":"Red Teaming Language Models with Language Models"},"references":{"count":15,"internal_anchors":2,"resolved_work":15,"sample":[{"cited_arxiv_id":"","doi":"","is_internal_anchor":false,"ref_index":1,"title":"In ICASSP 2019 - 2019 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pages 7345–7349","work_id":"ad751b1b-3615-45fc-bb1b-ff25cdd3c222","year":2019},{"cited_arxiv_id":"","doi":"","is_internal_anchor":false,"ref_index":2,"title":"In Advances in Neural Information Processing Systems, volume 33, pages 1877–1901","work_id":"2dc7b88c-3d81-4a2d-ba10-e9527d40e62d","year":1901},{"cited_arxiv_id":"2109.13916","doi":"","is_internal_anchor":true,"ref_index":3,"title":"Unsolved Problems in ML Safety","work_id":"a4b59a6c-1b80-4562-afd6-0e6d0126d3bc","year":2019},{"cited_arxiv_id":"1907.00456","doi":"","is_internal_anchor":false,"ref_index":4,"title":"Way Off-Policy Batch Deep Reinforcement Learning of Implicit Human Preferences in Dialog","work_id":"42fcaa3e-0409-481b-9dd5-9a2c3be8a383","year":2020},{"cited_arxiv_id":"","doi":"","is_internal_anchor":false,"ref_index":5,"title":"In Proceedings of the 2016 Conference on Empirical Methods in Natural Language Processing , pages 2122–2132, Austin, Texas","work_id":"f5ab341f-5b15-47d7-8f8f-a43b384b5be8","year":2016}],"snapshot_sha256":"d657602cd5a9cebddef5dd057608b12d3f1113aa668e147ceabed942b54f103a"},"source":{"id":"2202.03286","kind":"arxiv","version":1},"verdict":{"created_at":"2026-05-11T20:52:33.524394Z","id":"11e1f2f3-d1bb-4af8-ac57-f6d9722da3e3","model_set":{"reader":"grok-4.3"},"one_line_summary":"One language model can generate diverse test cases to automatically uncover tens of thousands of harmful behaviors, including offensive replies and privacy leaks, in a large target language model.","pipeline_version":"pith-pipeline@v0.9.0","pith_extraction_headline":"One language model generates test cases to automatically uncover tens of thousands of harmful behaviors in another language model.","strongest_claim":"we automatically find cases where a target LM behaves in a harmful way, by generating test cases (red teaming) using another LM... uncovering tens of thousands of offensive replies in a 280B parameter LM chatbot.","weakest_assumption":"The classifier trained to detect offensive content accurately identifies the relevant harms, and the LM-generated test cases are sufficiently diverse, difficult, and representative of real user interactions."}},"verdict_id":"11e1f2f3-d1bb-4af8-ac57-f6d9722da3e3"}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:81b3856f8ed5d0c4074e52b51173239013abeae29a30fd75ae1ccec2ccb07af5","target":"record","created_at":"2026-07-05T03:54:37Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"1ee4e6d94955492a585c6de94a1450e146e1c56b12e6b5d11fda444ee1f08518","cross_cats_sorted":["cs.AI","cs.CR","cs.LG"],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2022-02-07T15:22:17Z","title_canon_sha256":"bb80b8b6cea33a49c34048dc37f0a2a6ceb18513c8d0fdde3c17a667d6323388"},"schema_version":"1.0","source":{"id":"2202.03286","kind":"arxiv","version":1}},"canonical_sha256":"82478e6379b99ceee8540b446535fe8d65d0765e56e6bff9dbd0e44a46f4197a","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"82478e6379b99ceee8540b446535fe8d65d0765e56e6bff9dbd0e44a46f4197a","first_computed_at":"2026-07-05T03:54:37.615530Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T03:54:37.615530Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"9+MNayff/1mayqLmadyNBxRI8Gzme3+kX95n1xqRdeExFBiStHvYt484Df642h/02kVE8C/Iv2HxG1no9cU7Cw==","signature_status":"signed_v1","signed_at":"2026-07-05T03:54:37.615987Z","signed_message":"canonical_sha256_bytes"},"source_id":"2202.03286","source_kind":"arxiv","source_version":1}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:81b3856f8ed5d0c4074e52b51173239013abeae29a30fd75ae1ccec2ccb07af5","sha256:0e538566dee65066a44547e32bbe18bb39326b2f070871251c9be6b5d988602d"],"state_sha256":"f93b2228d74cebd29e057dc41bc19914e7ce666ba32ffc2ee921ff30a42e740f"}