{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:YI2TVPJK7HMEYCN46THOG3FTSF","short_pith_number":"pith:YI2TVPJK","schema_version":"1.0","canonical_sha256":"c2353abd2af9d84c09bcf4cee36cb391540506e5677d614f44b619b4fa52e62a","source":{"kind":"arxiv","id":"2404.08676","version":3},"attestation_state":"computed","paper":{"title":"ALERT: A Comprehensive Benchmark for Assessing Large Language Models' Safety through Red Teaming","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":["cs.CY","cs.LG"],"primary_cat":"cs.CL","authors_text":"Bo Li, Felix Friedrich, Huu Nguyen, Kristian Kersting, Patrick Schramowski, Roberto Navigli, Simone Tedeschi","submitted_at":"2024-04-06T15:01:47Z","abstract_excerpt":"When building Large Language Models (LLMs), it is paramount to bear safety in mind and protect them with guardrails. Indeed, LLMs should never generate content promoting or normalizing harmful, illegal, or unethical behavior that may contribute to harm to individuals or society. This principle applies to both normal and adversarial use. In response, we introduce ALERT, a large-scale benchmark to assess safety based on a novel fine-grained risk taxonomy. It is designed to evaluate the safety of LLMs through red teaming methodologies and consists of more than 45k instructions categorized using o"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2404.08676","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by-sa/4.0/","primary_cat":"cs.CL","submitted_at":"2024-04-06T15:01:47Z","cross_cats_sorted":["cs.CY","cs.LG"],"title_canon_sha256":"89884862badb8b047911b5045d87ca2b0b4d6381790b7bcd7561a7f9d29b74d3","abstract_canon_sha256":"fe9c93ac85c674fbf028c062688499f901656da0f6dea796a861359fbbfbd173"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:35:45.291878Z","signature_b64":"o73jUrLmn5kUJGx2igb5zB10zCz5kbFK+QzX0QBmVJmyBD1pFvmXIG3t5TDGVaQ4fELWitJep06JSJaSigNRAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"c2353abd2af9d84c09bcf4cee36cb391540506e5677d614f44b619b4fa52e62a","last_reissued_at":"2026-07-05T08:35:45.291420Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:35:45.291420Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"ALERT: A Comprehensive Benchmark for Assessing Large Language Models' Safety through Red Teaming","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":["cs.CY","cs.LG"],"primary_cat":"cs.CL","authors_text":"Bo Li, Felix Friedrich, Huu Nguyen, Kristian Kersting, Patrick Schramowski, Roberto Navigli, Simone Tedeschi","submitted_at":"2024-04-06T15:01:47Z","abstract_excerpt":"When building Large Language Models (LLMs), it is paramount to bear safety in mind and protect them with guardrails. Indeed, LLMs should never generate content promoting or normalizing harmful, illegal, or unethical behavior that may contribute to harm to individuals or society. This principle applies to both normal and adversarial use. In response, we introduce ALERT, a large-scale benchmark to assess safety based on a novel fine-grained risk taxonomy. It is designed to evaluate the safety of LLMs through red teaming methodologies and consists of more than 45k instructions categorized using o"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2404.08676","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2404.08676/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2404.08676","created_at":"2026-07-05T08:35:45.291485+00:00"},{"alias_kind":"arxiv_version","alias_value":"2404.08676v3","created_at":"2026-07-05T08:35:45.291485+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2404.08676","created_at":"2026-07-05T08:35:45.291485+00:00"},{"alias_kind":"pith_short_12","alias_value":"YI2TVPJK7HME","created_at":"2026-07-05T08:35:45.291485+00:00"},{"alias_kind":"pith_short_16","alias_value":"YI2TVPJK7HMEYCN4","created_at":"2026-07-05T08:35:45.291485+00:00"},{"alias_kind":"pith_short_8","alias_value":"YI2TVPJK","created_at":"2026-07-05T08:35:45.291485+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":9,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.19887","citing_title":"FinRED: An Expert-Guided Benchmark Generation and Evaluation Framework for Financial LLM Red-Teaming","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2606.09178","citing_title":"Culturally-Adapted Red-Teaming Across East and Southeast Asian Contexts: A Methodological and Comparative Analysis","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2606.06055","citing_title":"When Should Memory Stay Silent: Measuring Memory-Use Boundaries in Memory-Augmented Conversational Agents","ref_index":37,"is_internal_anchor":false},{"citing_arxiv_id":"2606.02530","citing_title":"SafeSteer: Localized On-Policy Distillation for Efficient Safety Alignment","ref_index":58,"is_internal_anchor":false},{"citing_arxiv_id":"2606.01322","citing_title":"TukaBench: A Culturally Grounded Jailbreak Benchmark for African Languages","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2601.17887","citing_title":"When Personalization Legitimizes Risks: Uncovering Safety Vulnerabilities in Personalized Dialogue Agents","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2510.21285","citing_title":"When Models Outthink Their Safety: Unveiling and Mitigating Self-Jailbreak in Large Reasoning Models","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2406.18495","citing_title":"WildGuard: Open One-Stop Moderation Tools for Safety Risks, Jailbreaks, and Refusals of LLMs","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2604.24826","citing_title":"A Comparative Evaluation of AI Agent Security Guardrails","ref_index":2,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/YI2TVPJK7HMEYCN46THOG3FTSF","json":"https://pith.science/pith/YI2TVPJK7HMEYCN46THOG3FTSF.json","graph_json":"https://pith.science/api/pith-number/YI2TVPJK7HMEYCN46THOG3FTSF/graph.json","events_json":"https://pith.science/api/pith-number/YI2TVPJK7HMEYCN46THOG3FTSF/events.json","paper":"https://pith.science/paper/YI2TVPJK"},"agent_actions":{"view_html":"https://pith.science/pith/YI2TVPJK7HMEYCN46THOG3FTSF","download_json":"https://pith.science/pith/YI2TVPJK7HMEYCN46THOG3FTSF.json","view_paper":"https://pith.science/paper/YI2TVPJK","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2404.08676&json=true","fetch_graph":"https://pith.science/api/pith-number/YI2TVPJK7HMEYCN46THOG3FTSF/graph.json","fetch_events":"https://pith.science/api/pith-number/YI2TVPJK7HMEYCN46THOG3FTSF/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/YI2TVPJK7HMEYCN46THOG3FTSF/action/timestamp_anchor","attest_storage":"https://pith.science/pith/YI2TVPJK7HMEYCN46THOG3FTSF/action/storage_attestation","attest_author":"https://pith.science/pith/YI2TVPJK7HMEYCN46THOG3FTSF/action/author_attestation","sign_citation":"https://pith.science/pith/YI2TVPJK7HMEYCN46THOG3FTSF/action/citation_signature","submit_replication":"https://pith.science/pith/YI2TVPJK7HMEYCN46THOG3FTSF/action/replication_record"}},"created_at":"2026-07-05T08:35:45.291485+00:00","updated_at":"2026-07-05T08:35:45.291485+00:00"}