{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:ESFIJHBOUM33S4JE5L3NNDOTT2","short_pith_number":"pith:ESFIJHBO","schema_version":"1.0","canonical_sha256":"248a849c2ea337b97124eaf6d68dd39e901aca8e27f7688af29ba10a22d37925","source":{"kind":"arxiv","id":"2503.01742","version":2},"attestation_state":"computed","paper":{"title":"Building Safe GenAI Applications: An End-to-End Overview of Red Teaming for Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Akshay Gupta, Alberto Purpura, Andy Luo, Jesse Zymet, Melissa Kazemi Rad, Mohammad Shahed Sorower, Sahil Wadhwa, Swapnil Shinde","submitted_at":"2025-03-03T17:04:22Z","abstract_excerpt":"The rapid growth of Large Language Models (LLMs) presents significant privacy, security, and ethical concerns. While much research has proposed methods for defending LLM systems against misuse by malicious actors, researchers have recently complemented these efforts with an offensive approach that involves red teaming, i.e., proactively attacking LLMs with the purpose of identifying their vulnerabilities. This paper provides a concise and practical overview of the LLM red teaming literature, structured so as to describe a multi-component system end-to-end. To motivate red teaming we survey the"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2503.01742","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-03-03T17:04:22Z","cross_cats_sorted":[],"title_canon_sha256":"06cc47496fbe7bac25233f1aa333d8c90988f117c0c257f7572cd772ef46dfc3","abstract_canon_sha256":"8c6dbe9821573a8e6b79c12b850ff478e02e25fd4cbf0f095774a1923ea501b1"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:24:32.887292Z","signature_b64":"LNaoNLp4y85IWXiT3cOCYM0YEeKQXWRNOrBsETe5Ef64gl+THWr0pSHw8W3+B04Cn6nTu33cS1Ll/bxmbOTeCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"248a849c2ea337b97124eaf6d68dd39e901aca8e27f7688af29ba10a22d37925","last_reissued_at":"2026-07-05T10:24:32.886367Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:24:32.886367Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Building Safe GenAI Applications: An End-to-End Overview of Red Teaming for Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Akshay Gupta, Alberto Purpura, Andy Luo, Jesse Zymet, Melissa Kazemi Rad, Mohammad Shahed Sorower, Sahil Wadhwa, Swapnil Shinde","submitted_at":"2025-03-03T17:04:22Z","abstract_excerpt":"The rapid growth of Large Language Models (LLMs) presents significant privacy, security, and ethical concerns. While much research has proposed methods for defending LLM systems against misuse by malicious actors, researchers have recently complemented these efforts with an offensive approach that involves red teaming, i.e., proactively attacking LLMs with the purpose of identifying their vulnerabilities. This paper provides a concise and practical overview of the LLM red teaming literature, structured so as to describe a multi-component system end-to-end. To motivate red teaming we survey the"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2503.01742","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2503.01742/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2503.01742","created_at":"2026-07-05T10:24:32.886500+00:00"},{"alias_kind":"arxiv_version","alias_value":"2503.01742v2","created_at":"2026-07-05T10:24:32.886500+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2503.01742","created_at":"2026-07-05T10:24:32.886500+00:00"},{"alias_kind":"pith_short_12","alias_value":"ESFIJHBOUM33","created_at":"2026-07-05T10:24:32.886500+00:00"},{"alias_kind":"pith_short_16","alias_value":"ESFIJHBOUM33S4JE","created_at":"2026-07-05T10:24:32.886500+00:00"},{"alias_kind":"pith_short_8","alias_value":"ESFIJHBO","created_at":"2026-07-05T10:24:32.886500+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.30096","citing_title":"How Reliable Are AI Attackers Against a Fixed Vulnerable Target? A 400-Run Empirical Study of LLM Penetration Testing Consistency","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2509.12626","citing_title":"DoubleAgents: Human-Agent Alignment in a Socially Embedded Workflow","ref_index":33,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ESFIJHBOUM33S4JE5L3NNDOTT2","json":"https://pith.science/pith/ESFIJHBOUM33S4JE5L3NNDOTT2.json","graph_json":"https://pith.science/api/pith-number/ESFIJHBOUM33S4JE5L3NNDOTT2/graph.json","events_json":"https://pith.science/api/pith-number/ESFIJHBOUM33S4JE5L3NNDOTT2/events.json","paper":"https://pith.science/paper/ESFIJHBO"},"agent_actions":{"view_html":"https://pith.science/pith/ESFIJHBOUM33S4JE5L3NNDOTT2","download_json":"https://pith.science/pith/ESFIJHBOUM33S4JE5L3NNDOTT2.json","view_paper":"https://pith.science/paper/ESFIJHBO","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2503.01742&json=true","fetch_graph":"https://pith.science/api/pith-number/ESFIJHBOUM33S4JE5L3NNDOTT2/graph.json","fetch_events":"https://pith.science/api/pith-number/ESFIJHBOUM33S4JE5L3NNDOTT2/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ESFIJHBOUM33S4JE5L3NNDOTT2/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ESFIJHBOUM33S4JE5L3NNDOTT2/action/storage_attestation","attest_author":"https://pith.science/pith/ESFIJHBOUM33S4JE5L3NNDOTT2/action/author_attestation","sign_citation":"https://pith.science/pith/ESFIJHBOUM33S4JE5L3NNDOTT2/action/citation_signature","submit_replication":"https://pith.science/pith/ESFIJHBOUM33S4JE5L3NNDOTT2/action/replication_record"}},"created_at":"2026-07-05T10:24:32.886500+00:00","updated_at":"2026-07-05T10:24:32.886500+00:00"}