{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:S34C3IET5XLMCHQQ3T7QUOPAVF","short_pith_number":"pith:S34C3IET","schema_version":"1.0","canonical_sha256":"96f82da093edd6c11e10dcff0a39e0a972e3fc70f94bc95d180c36ff1fd1e947","source":{"kind":"arxiv","id":"2308.13387","version":2},"attestation_state":"computed","paper":{"title":"Do-Not-Answer: A Dataset for Evaluating Safeguards in LLMs","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Haonan Li, Preslav Nakov, Timothy Baldwin, Xudong Han, Yuxia Wang","submitted_at":"2023-08-25T14:02:12Z","abstract_excerpt":"With the rapid evolution of large language models (LLMs), new and hard-to-predict harmful capabilities are emerging. This requires developers to be able to identify risks through the evaluation of \"dangerous capabilities\" in order to responsibly deploy LLMs. In this work, we collect the first open-source dataset to evaluate safeguards in LLMs, and deploy safer open-source LLMs at a low cost. Our dataset is curated and filtered to consist only of instructions that responsible language models should not follow. We annotate and assess the responses of six popular LLMs to these instructions. Based"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2308.13387","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2023-08-25T14:02:12Z","cross_cats_sorted":[],"title_canon_sha256":"df8ae39571cb6863844b9e2dc6f67898fe694db39971583045b660dc8ad1e0e5","abstract_canon_sha256":"2bbceec3d2717410bcfff9b13ae4625f6123f3b1ab5e94df1fb549ca71d50952"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:47:27.136020Z","signature_b64":"tCmtL43gqEhr2DPU4OzlKXzce3Zd5tFmX+cU1k5cKEEC2bzv8bvWU5FCz8T+aAeJ+mWy0/Sr0em0heJPh7b3AA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"96f82da093edd6c11e10dcff0a39e0a972e3fc70f94bc95d180c36ff1fd1e947","last_reissued_at":"2026-07-05T06:47:27.135543Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:47:27.135543Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Do-Not-Answer: A Dataset for Evaluating Safeguards in LLMs","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Haonan Li, Preslav Nakov, Timothy Baldwin, Xudong Han, Yuxia Wang","submitted_at":"2023-08-25T14:02:12Z","abstract_excerpt":"With the rapid evolution of large language models (LLMs), new and hard-to-predict harmful capabilities are emerging. This requires developers to be able to identify risks through the evaluation of \"dangerous capabilities\" in order to responsibly deploy LLMs. In this work, we collect the first open-source dataset to evaluate safeguards in LLMs, and deploy safer open-source LLMs at a low cost. Our dataset is curated and filtered to consist only of instructions that responsible language models should not follow. We annotate and assess the responses of six popular LLMs to these instructions. Based"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2308.13387","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2308.13387/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2308.13387","created_at":"2026-07-05T06:47:27.135601+00:00"},{"alias_kind":"arxiv_version","alias_value":"2308.13387v2","created_at":"2026-07-05T06:47:27.135601+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2308.13387","created_at":"2026-07-05T06:47:27.135601+00:00"},{"alias_kind":"pith_short_12","alias_value":"S34C3IET5XLM","created_at":"2026-07-05T06:47:27.135601+00:00"},{"alias_kind":"pith_short_16","alias_value":"S34C3IET5XLMCHQQ","created_at":"2026-07-05T06:47:27.135601+00:00"},{"alias_kind":"pith_short_8","alias_value":"S34C3IET","created_at":"2026-07-05T06:47:27.135601+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":26,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.25782","citing_title":"Do Encoders Suffice? A Systematic Comparison of Encoder and Decoder Safety Judges for LLM Adversarial Evaluation","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2606.19887","citing_title":"FinRED: An Expert-Guided Benchmark Generation and Evaluation Framework for Financial LLM Red-Teaming","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2606.19594","citing_title":"Unsupervised Causal Abstractions Discovery","ref_index":84,"is_internal_anchor":false},{"citing_arxiv_id":"2606.08044","citing_title":"When Behavioral Safety Evaluation Fails: A Representation-Level Perspective","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2606.20626","citing_title":"Efficient Safety Benchmarking via Item Response Theory","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2605.24414","citing_title":"JT-SAFE-V2: Safety-by-Design Foundation Model with World-Context Data","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2606.29171","citing_title":"Symbolic Mechanistic Data Attribution: Tracing Training Influence to Learned Behavioral Policies","ref_index":62,"is_internal_anchor":false},{"citing_arxiv_id":"2605.24817","citing_title":"RouteScan: A Non-Intrusive Approach to Auditing MoE LLMs Safety via Expert Routing Telemetry","ref_index":49,"is_internal_anchor":false},{"citing_arxiv_id":"2606.00875","citing_title":"IDEAFix: Evaluation Framework for Creative Defixation Prompting in LLMs","ref_index":50,"is_internal_anchor":false},{"citing_arxiv_id":"2606.21296","citing_title":"Discriminatory Compliance: How LLMs Answer Queries from Protected Groups","ref_index":100,"is_internal_anchor":false},{"citing_arxiv_id":"2602.07340","citing_title":"Revisiting Robustness for LLM Safety Alignment via Selective Geometry Control","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2605.20654","citing_title":"REFLECTOR: Internalizing Step-wise Reflection against Indirect Jailbreak","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2507.00432","citing_title":"Does Math Reasoning Improve General LLM Capabilities? Understanding Transferability of LLM Reasoning","ref_index":234,"is_internal_anchor":false},{"citing_arxiv_id":"2401.05561","citing_title":"TrustLLM: Trustworthiness in Large Language Models","ref_index":73,"is_internal_anchor":false},{"citing_arxiv_id":"2310.02446","citing_title":"Low-Resource Languages Jailbreak GPT-4","ref_index":51,"is_internal_anchor":false},{"citing_arxiv_id":"2407.04295","citing_title":"Jailbreak Attacks and Defenses Against Large Language Models: A Survey","ref_index":96,"is_internal_anchor":false},{"citing_arxiv_id":"2604.01444","citing_title":"Cooking Up Risks: Benchmarking and Reducing Food Safety Risks in Large Language Models","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10639","citing_title":"Navigating the Sea of LLM Evaluation: Investigating Bias in Toxicity Benchmarks","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06213","citing_title":"Beyond Fixed Benchmarks and Worst-Case Attacks: Dynamic Boundary Evaluation for Language Models","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01687","citing_title":"MultiBreak: A Scalable and Diverse Multi-turn Jailbreak Benchmark for Evaluating LLM Safety","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2604.07655","citing_title":"Guardian-as-an-Advisor: Advancing Next-Generation Guardian Models for Trustworthy LLMs","ref_index":80,"is_internal_anchor":false},{"citing_arxiv_id":"2401.02954","citing_title":"DeepSeek LLM: Scaling Open-Source Language Models with Longtermism","ref_index":151,"is_internal_anchor":false},{"citing_arxiv_id":"2405.04434","citing_title":"DeepSeek-V2: A Strong, Economical, and Efficient Mixture-of-Experts Language Model","ref_index":149,"is_internal_anchor":false},{"citing_arxiv_id":"2604.07754","citing_title":"The Art of (Mis)alignment: How Fine-Tuning Methods Effectively Misalign and Realign LLMs in Post-Training","ref_index":60,"is_internal_anchor":false},{"citing_arxiv_id":"2604.21152","citing_title":"Dialect vs Demographics: Quantifying LLM Bias from Implicit Linguistic Signals vs. Explicit User Profiles","ref_index":24,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/S34C3IET5XLMCHQQ3T7QUOPAVF","json":"https://pith.science/pith/S34C3IET5XLMCHQQ3T7QUOPAVF.json","graph_json":"https://pith.science/api/pith-number/S34C3IET5XLMCHQQ3T7QUOPAVF/graph.json","events_json":"https://pith.science/api/pith-number/S34C3IET5XLMCHQQ3T7QUOPAVF/events.json","paper":"https://pith.science/paper/S34C3IET"},"agent_actions":{"view_html":"https://pith.science/pith/S34C3IET5XLMCHQQ3T7QUOPAVF","download_json":"https://pith.science/pith/S34C3IET5XLMCHQQ3T7QUOPAVF.json","view_paper":"https://pith.science/paper/S34C3IET","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2308.13387&json=true","fetch_graph":"https://pith.science/api/pith-number/S34C3IET5XLMCHQQ3T7QUOPAVF/graph.json","fetch_events":"https://pith.science/api/pith-number/S34C3IET5XLMCHQQ3T7QUOPAVF/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/S34C3IET5XLMCHQQ3T7QUOPAVF/action/timestamp_anchor","attest_storage":"https://pith.science/pith/S34C3IET5XLMCHQQ3T7QUOPAVF/action/storage_attestation","attest_author":"https://pith.science/pith/S34C3IET5XLMCHQQ3T7QUOPAVF/action/author_attestation","sign_citation":"https://pith.science/pith/S34C3IET5XLMCHQQ3T7QUOPAVF/action/citation_signature","submit_replication":"https://pith.science/pith/S34C3IET5XLMCHQQ3T7QUOPAVF/action/replication_record"}},"created_at":"2026-07-05T06:47:27.135601+00:00","updated_at":"2026-07-05T06:47:27.135601+00:00"}