{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:JINEQKUTW5F5AHAQIFE5CD6CE3","short_pith_number":"pith:JINEQKUT","schema_version":"1.0","canonical_sha256":"4a1a482a93b74bd01c104149d10fc226ceacfd6571e910974c6e4cd3593f0a41","source":{"kind":"arxiv","id":"2410.01606","version":1},"attestation_state":"computed","paper":{"title":"Automated Red Teaming with GOAT: the Generative Offensive Agent Tester","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Aaron Grattafiori, Cristian Canton Ferrer, Erik Brinkman, Hailey Nguyen, Ivan Evtimov, Joanna Bitton, Joe Li, Krithika Iyer, Maya Pavlova, Vitor Albiero","submitted_at":"2024-10-02T14:47:05Z","abstract_excerpt":"Red teaming assesses how large language models (LLMs) can produce content that violates norms, policies, and rules set during their safety training. However, most existing automated methods in the literature are not representative of the way humans tend to interact with AI models. Common users of AI models may not have advanced knowledge of adversarial machine learning methods or access to model internals, and they do not spend a lot of time crafting a single highly effective adversarial prompt. Instead, they are likely to make use of techniques commonly shared online and exploit the multiturn"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.01606","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-10-02T14:47:05Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"f7302572c7dbda419650ff079ca30f147a2a307b3722c5129d2342c0b8e4611f","abstract_canon_sha256":"05ff4c7e73aeafab5a10bc8ec42a4bf7081b3f04de9af2d05614a9c9d79b28f0"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:14:49.896379Z","signature_b64":"1tg/VM6cIYly1uJ1qm5Kic4idtmnCxt1bsnXRABaJYI43sEQTJW94NfGl4nDRrwwVJffGMB2fLgdiai+A1dvDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"4a1a482a93b74bd01c104149d10fc226ceacfd6571e910974c6e4cd3593f0a41","last_reissued_at":"2026-07-05T09:14:49.895931Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:14:49.895931Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Automated Red Teaming with GOAT: the Generative Offensive Agent Tester","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Aaron Grattafiori, Cristian Canton Ferrer, Erik Brinkman, Hailey Nguyen, Ivan Evtimov, Joanna Bitton, Joe Li, Krithika Iyer, Maya Pavlova, Vitor Albiero","submitted_at":"2024-10-02T14:47:05Z","abstract_excerpt":"Red teaming assesses how large language models (LLMs) can produce content that violates norms, policies, and rules set during their safety training. However, most existing automated methods in the literature are not representative of the way humans tend to interact with AI models. Common users of AI models may not have advanced knowledge of adversarial machine learning methods or access to model internals, and they do not spend a lot of time crafting a single highly effective adversarial prompt. Instead, they are likely to make use of techniques commonly shared online and exploit the multiturn"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.01606","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.01606/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.01606","created_at":"2026-07-05T09:14:49.895989+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.01606v1","created_at":"2026-07-05T09:14:49.895989+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.01606","created_at":"2026-07-05T09:14:49.895989+00:00"},{"alias_kind":"pith_short_12","alias_value":"JINEQKUTW5F5","created_at":"2026-07-05T09:14:49.895989+00:00"},{"alias_kind":"pith_short_16","alias_value":"JINEQKUTW5F5AHAQ","created_at":"2026-07-05T09:14:49.895989+00:00"},{"alias_kind":"pith_short_8","alias_value":"JINEQKUT","created_at":"2026-07-05T09:14:49.895989+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.25476","citing_title":"A Red Teaming Framework for Large Language Models: A Case Study on Faithfulness Evaluation","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2606.12737","citing_title":"PI-Hunter: Automated Red-Teaming for Exposing and Localizing Prompt Injections","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05630","citing_title":"One Turn Too Late: Response-Aware Defense Against Hidden Malicious Intent in Multi-Turn Dialogue","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05630","citing_title":"One Turn Too Late: Response-Aware Defense Against Hidden Malicious Intent in Multi-Turn Dialogue","ref_index":22,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/JINEQKUTW5F5AHAQIFE5CD6CE3","json":"https://pith.science/pith/JINEQKUTW5F5AHAQIFE5CD6CE3.json","graph_json":"https://pith.science/api/pith-number/JINEQKUTW5F5AHAQIFE5CD6CE3/graph.json","events_json":"https://pith.science/api/pith-number/JINEQKUTW5F5AHAQIFE5CD6CE3/events.json","paper":"https://pith.science/paper/JINEQKUT"},"agent_actions":{"view_html":"https://pith.science/pith/JINEQKUTW5F5AHAQIFE5CD6CE3","download_json":"https://pith.science/pith/JINEQKUTW5F5AHAQIFE5CD6CE3.json","view_paper":"https://pith.science/paper/JINEQKUT","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.01606&json=true","fetch_graph":"https://pith.science/api/pith-number/JINEQKUTW5F5AHAQIFE5CD6CE3/graph.json","fetch_events":"https://pith.science/api/pith-number/JINEQKUTW5F5AHAQIFE5CD6CE3/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/JINEQKUTW5F5AHAQIFE5CD6CE3/action/timestamp_anchor","attest_storage":"https://pith.science/pith/JINEQKUTW5F5AHAQIFE5CD6CE3/action/storage_attestation","attest_author":"https://pith.science/pith/JINEQKUTW5F5AHAQIFE5CD6CE3/action/author_attestation","sign_citation":"https://pith.science/pith/JINEQKUTW5F5AHAQIFE5CD6CE3/action/citation_signature","submit_replication":"https://pith.science/pith/JINEQKUTW5F5AHAQIFE5CD6CE3/action/replication_record"}},"created_at":"2026-07-05T09:14:49.895989+00:00","updated_at":"2026-07-05T09:14:49.895989+00:00"}