{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:2T47MLW5K5Q4JEGDKJLFJBGT6H","short_pith_number":"pith:2T47MLW5","schema_version":"1.0","canonical_sha256":"d4f9f62edd5761c490c352565484d3f1c5c4419eff6e5152f49be78e76b3a88a","source":{"kind":"arxiv","id":"2410.02916","version":3},"attestation_state":"computed","paper":{"title":"LLM Safeguard is a Double-Edged Sword: Exploiting False Positives for Denial-of-Service Attacks","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CR","authors_text":"Qingzhao Zhang, Ziyang Xiong, Z. Morley Mao","submitted_at":"2024-10-03T19:07:53Z","abstract_excerpt":"Safety is a paramount concern for large language models (LLMs) in open deployment, motivating the development of safeguard methods that enforce ethical and responsible use through safety alignment or guardrail mechanisms. Jailbreak attacks that exploit the \\emph{false negatives} of safeguard methods have emerged as a prominent research focus in the field of LLM security. However, we found that the malicious attackers could also exploit false positives of safeguards, i.e., fooling the safeguard model to block safe content mistakenly, leading to a denial-of-service (DoS) affecting LLM users. To "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.02916","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CR","submitted_at":"2024-10-03T19:07:53Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"394721fd259cf2a2aed101b4f830bb051b7f6e1033971934557b490e1d7cd43d","abstract_canon_sha256":"31cba27f7a52d6e813eef65ae4efd86e166f88aaf87ccd4e2996e06f571f8049"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:46:38.700433Z","signature_b64":"uL8HtvGrK/4H9+Guo1ibHcPROeqDtPNQqnByAVr9UQFL+zegcDITw0ELRg+hRnu4XDARgYxL/6snc/lHWloJAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"d4f9f62edd5761c490c352565484d3f1c5c4419eff6e5152f49be78e76b3a88a","last_reissued_at":"2026-07-05T10:46:38.699998Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:46:38.699998Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"LLM Safeguard is a Double-Edged Sword: Exploiting False Positives for Denial-of-Service Attacks","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CR","authors_text":"Qingzhao Zhang, Ziyang Xiong, Z. Morley Mao","submitted_at":"2024-10-03T19:07:53Z","abstract_excerpt":"Safety is a paramount concern for large language models (LLMs) in open deployment, motivating the development of safeguard methods that enforce ethical and responsible use through safety alignment or guardrail mechanisms. Jailbreak attacks that exploit the \\emph{false negatives} of safeguard methods have emerged as a prominent research focus in the field of LLM security. However, we found that the malicious attackers could also exploit false positives of safeguards, i.e., fooling the safeguard model to block safe content mistakenly, leading to a denial-of-service (DoS) affecting LLM users. To "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.02916","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.02916/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.02916","created_at":"2026-07-05T10:46:38.700048+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.02916v3","created_at":"2026-07-05T10:46:38.700048+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.02916","created_at":"2026-07-05T10:46:38.700048+00:00"},{"alias_kind":"pith_short_12","alias_value":"2T47MLW5K5Q4","created_at":"2026-07-05T10:46:38.700048+00:00"},{"alias_kind":"pith_short_16","alias_value":"2T47MLW5K5Q4JEGD","created_at":"2026-07-05T10:46:38.700048+00:00"},{"alias_kind":"pith_short_8","alias_value":"2T47MLW5","created_at":"2026-07-05T10:46:38.700048+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.12813","citing_title":"REALISTA: Realistic Latent Adversarial Attacks that Elicit LLM Hallucinations","ref_index":194,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/2T47MLW5K5Q4JEGDKJLFJBGT6H","json":"https://pith.science/pith/2T47MLW5K5Q4JEGDKJLFJBGT6H.json","graph_json":"https://pith.science/api/pith-number/2T47MLW5K5Q4JEGDKJLFJBGT6H/graph.json","events_json":"https://pith.science/api/pith-number/2T47MLW5K5Q4JEGDKJLFJBGT6H/events.json","paper":"https://pith.science/paper/2T47MLW5"},"agent_actions":{"view_html":"https://pith.science/pith/2T47MLW5K5Q4JEGDKJLFJBGT6H","download_json":"https://pith.science/pith/2T47MLW5K5Q4JEGDKJLFJBGT6H.json","view_paper":"https://pith.science/paper/2T47MLW5","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.02916&json=true","fetch_graph":"https://pith.science/api/pith-number/2T47MLW5K5Q4JEGDKJLFJBGT6H/graph.json","fetch_events":"https://pith.science/api/pith-number/2T47MLW5K5Q4JEGDKJLFJBGT6H/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/2T47MLW5K5Q4JEGDKJLFJBGT6H/action/timestamp_anchor","attest_storage":"https://pith.science/pith/2T47MLW5K5Q4JEGDKJLFJBGT6H/action/storage_attestation","attest_author":"https://pith.science/pith/2T47MLW5K5Q4JEGDKJLFJBGT6H/action/author_attestation","sign_citation":"https://pith.science/pith/2T47MLW5K5Q4JEGDKJLFJBGT6H/action/citation_signature","submit_replication":"https://pith.science/pith/2T47MLW5K5Q4JEGDKJLFJBGT6H/action/replication_record"}},"created_at":"2026-07-05T10:46:38.700048+00:00","updated_at":"2026-07-05T10:46:38.700048+00:00"}