{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:IPWNDMFUIL4TF5OPJY65U6WHTZ","short_pith_number":"pith:IPWNDMFU","schema_version":"1.0","canonical_sha256":"43ecd1b0b442f932f5cf4e3dda7ac79e6e48e21df38614bc293ea1509fdcf397","source":{"kind":"arxiv","id":"2503.16431","version":1},"attestation_state":"computed","paper":{"title":"OpenAI's Approach to External Red Teaming for AI Models and Systems","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CR","cs.HC"],"primary_cat":"cs.CY","authors_text":"Lama Ahmad, Michael Lampe, Pamela Mishkin, Sandhini Agarwal","submitted_at":"2025-01-24T20:54:48Z","abstract_excerpt":"Red teaming has emerged as a critical practice in assessing the possible risks of AI models and systems. It aids in the discovery of novel risks, stress testing possible gaps in existing mitigations, enriching existing quantitative safety metrics, facilitating the creation of new safety measurements, and enhancing public trust and the legitimacy of AI risk assessments. This white paper describes OpenAI's work to date in external red teaming and draws some more general conclusions from this work. We describe the design considerations underpinning external red teaming, which include: selecting c"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2503.16431","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CY","submitted_at":"2025-01-24T20:54:48Z","cross_cats_sorted":["cs.AI","cs.CR","cs.HC"],"title_canon_sha256":"24caac34a913da7e6c63fea34f6948d9df0c0432372d38a1f182335726f92b86","abstract_canon_sha256":"eed03a489a386c71d0b613ac9f628808155ac037c57c9c5b2d0c1601c073d8b8"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:36:40.534891Z","signature_b64":"Ng/f4fx6+xpzLYQREZ5Fohw5JZgh9OaP7u92X6nN2EooXGNYD/FQk/NBJdNIpaaZ+UoTUbcxA3aRF0Xsf2yWDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"43ecd1b0b442f932f5cf4e3dda7ac79e6e48e21df38614bc293ea1509fdcf397","last_reissued_at":"2026-07-05T10:36:40.534350Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:36:40.534350Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"OpenAI's Approach to External Red Teaming for AI Models and Systems","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CR","cs.HC"],"primary_cat":"cs.CY","authors_text":"Lama Ahmad, Michael Lampe, Pamela Mishkin, Sandhini Agarwal","submitted_at":"2025-01-24T20:54:48Z","abstract_excerpt":"Red teaming has emerged as a critical practice in assessing the possible risks of AI models and systems. It aids in the discovery of novel risks, stress testing possible gaps in existing mitigations, enriching existing quantitative safety metrics, facilitating the creation of new safety measurements, and enhancing public trust and the legitimacy of AI risk assessments. This white paper describes OpenAI's work to date in external red teaming and draws some more general conclusions from this work. We describe the design considerations underpinning external red teaming, which include: selecting c"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2503.16431","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2503.16431/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2503.16431","created_at":"2026-07-05T10:36:40.534418+00:00"},{"alias_kind":"arxiv_version","alias_value":"2503.16431v1","created_at":"2026-07-05T10:36:40.534418+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2503.16431","created_at":"2026-07-05T10:36:40.534418+00:00"},{"alias_kind":"pith_short_12","alias_value":"IPWNDMFUIL4T","created_at":"2026-07-05T10:36:40.534418+00:00"},{"alias_kind":"pith_short_16","alias_value":"IPWNDMFUIL4TF5OP","created_at":"2026-07-05T10:36:40.534418+00:00"},{"alias_kind":"pith_short_8","alias_value":"IPWNDMFU","created_at":"2026-07-05T10:36:40.534418+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.30661","citing_title":"Understanding Censorship in Large Language Models: From Mechanisms to Governance","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2510.07239","citing_title":"Red-Bandit: Test-Time Adaptation for LLM Red-Teaming via Bandit-Guided LoRA Experts","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2605.16198","citing_title":"Formal Methods Meet LLMs: Auditing, Monitoring, and Intervention for Compliance of Advanced AI Systems","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2604.28157","citing_title":"FlashRT: Towards Computationally and Memory Efficient Red-Teaming for Prompt Injection and Knowledge Corruption","ref_index":45,"is_internal_anchor":false},{"citing_arxiv_id":"2604.21864","citing_title":"FAccT-Checked: A Narrative Review of Authority Reconfigurations and Retention in AI-Mediated Journalism","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2604.20833","citing_title":"AVISE: Framework for Evaluating the Security of AI Systems","ref_index":34,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/IPWNDMFUIL4TF5OPJY65U6WHTZ","json":"https://pith.science/pith/IPWNDMFUIL4TF5OPJY65U6WHTZ.json","graph_json":"https://pith.science/api/pith-number/IPWNDMFUIL4TF5OPJY65U6WHTZ/graph.json","events_json":"https://pith.science/api/pith-number/IPWNDMFUIL4TF5OPJY65U6WHTZ/events.json","paper":"https://pith.science/paper/IPWNDMFU"},"agent_actions":{"view_html":"https://pith.science/pith/IPWNDMFUIL4TF5OPJY65U6WHTZ","download_json":"https://pith.science/pith/IPWNDMFUIL4TF5OPJY65U6WHTZ.json","view_paper":"https://pith.science/paper/IPWNDMFU","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2503.16431&json=true","fetch_graph":"https://pith.science/api/pith-number/IPWNDMFUIL4TF5OPJY65U6WHTZ/graph.json","fetch_events":"https://pith.science/api/pith-number/IPWNDMFUIL4TF5OPJY65U6WHTZ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/IPWNDMFUIL4TF5OPJY65U6WHTZ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/IPWNDMFUIL4TF5OPJY65U6WHTZ/action/storage_attestation","attest_author":"https://pith.science/pith/IPWNDMFUIL4TF5OPJY65U6WHTZ/action/author_attestation","sign_citation":"https://pith.science/pith/IPWNDMFUIL4TF5OPJY65U6WHTZ/action/citation_signature","submit_replication":"https://pith.science/pith/IPWNDMFUIL4TF5OPJY65U6WHTZ/action/replication_record"}},"created_at":"2026-07-05T10:36:40.534418+00:00","updated_at":"2026-07-05T10:36:40.534418+00:00"}