{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:JRXAPAQO5WZDCAD2OU2HMQKSQX","short_pith_number":"pith:JRXAPAQO","schema_version":"1.0","canonical_sha256":"4c6e07820eedb231007a753476415285ee906ad8df8b3add5f6305897576a329","source":{"kind":"arxiv","id":"2405.05418","version":2},"attestation_state":"computed","paper":{"title":"Mitigating Exaggerated Safety in Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Ruchi Bhalani, Ruchira Ray","submitted_at":"2024-05-08T20:39:54Z","abstract_excerpt":"As the popularity of Large Language Models (LLMs) grow, combining model safety with utility becomes increasingly important. The challenge is making sure that LLMs can recognize and decline dangerous prompts without sacrificing their ability to be helpful. The problem of \"exaggerated safety\" demonstrates how difficult this can be. To reduce excessive safety behaviours -- which was discovered to be 26.1% of safe prompts being misclassified as dangerous and refused -- we use a combination of XSTest dataset prompts as well as interactive, contextual, and few-shot prompting to examine the decision "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2405.05418","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-05-08T20:39:54Z","cross_cats_sorted":[],"title_canon_sha256":"493d3ee2b04bc76234edebebe7b86c822ea9cc3429ee2e15c291375d4f2dd19f","abstract_canon_sha256":"3b66fca70afcc34a3147c4d7debbbb2354d6bd91ced088f24a6db59c3e6f1226"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:00:37.183899Z","signature_b64":"jk5eaFEyWOQztrZSeV/UakRBOyxzxuufOVuNng5zuxBiGZNwVdH2/3PVMIf5QR4uugwhb0CWLyw+ZK8oIYtCAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"4c6e07820eedb231007a753476415285ee906ad8df8b3add5f6305897576a329","last_reissued_at":"2026-07-05T09:00:37.183488Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:00:37.183488Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Mitigating Exaggerated Safety in Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Ruchi Bhalani, Ruchira Ray","submitted_at":"2024-05-08T20:39:54Z","abstract_excerpt":"As the popularity of Large Language Models (LLMs) grow, combining model safety with utility becomes increasingly important. The challenge is making sure that LLMs can recognize and decline dangerous prompts without sacrificing their ability to be helpful. The problem of \"exaggerated safety\" demonstrates how difficult this can be. To reduce excessive safety behaviours -- which was discovered to be 26.1% of safe prompts being misclassified as dangerous and refused -- we use a combination of XSTest dataset prompts as well as interactive, contextual, and few-shot prompting to examine the decision "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2405.05418","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2405.05418/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2405.05418","created_at":"2026-07-05T09:00:37.183541+00:00"},{"alias_kind":"arxiv_version","alias_value":"2405.05418v2","created_at":"2026-07-05T09:00:37.183541+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2405.05418","created_at":"2026-07-05T09:00:37.183541+00:00"},{"alias_kind":"pith_short_12","alias_value":"JRXAPAQO5WZD","created_at":"2026-07-05T09:00:37.183541+00:00"},{"alias_kind":"pith_short_16","alias_value":"JRXAPAQO5WZDCAD2","created_at":"2026-07-05T09:00:37.183541+00:00"},{"alias_kind":"pith_short_8","alias_value":"JRXAPAQO","created_at":"2026-07-05T09:00:37.183541+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.24154","citing_title":"Palette: A Modular, Controllable, and Efficient Framework for On-demand Authorized Safety Alignment Relaxation in LLMs","ref_index":54,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/JRXAPAQO5WZDCAD2OU2HMQKSQX","json":"https://pith.science/pith/JRXAPAQO5WZDCAD2OU2HMQKSQX.json","graph_json":"https://pith.science/api/pith-number/JRXAPAQO5WZDCAD2OU2HMQKSQX/graph.json","events_json":"https://pith.science/api/pith-number/JRXAPAQO5WZDCAD2OU2HMQKSQX/events.json","paper":"https://pith.science/paper/JRXAPAQO"},"agent_actions":{"view_html":"https://pith.science/pith/JRXAPAQO5WZDCAD2OU2HMQKSQX","download_json":"https://pith.science/pith/JRXAPAQO5WZDCAD2OU2HMQKSQX.json","view_paper":"https://pith.science/paper/JRXAPAQO","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2405.05418&json=true","fetch_graph":"https://pith.science/api/pith-number/JRXAPAQO5WZDCAD2OU2HMQKSQX/graph.json","fetch_events":"https://pith.science/api/pith-number/JRXAPAQO5WZDCAD2OU2HMQKSQX/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/JRXAPAQO5WZDCAD2OU2HMQKSQX/action/timestamp_anchor","attest_storage":"https://pith.science/pith/JRXAPAQO5WZDCAD2OU2HMQKSQX/action/storage_attestation","attest_author":"https://pith.science/pith/JRXAPAQO5WZDCAD2OU2HMQKSQX/action/author_attestation","sign_citation":"https://pith.science/pith/JRXAPAQO5WZDCAD2OU2HMQKSQX/action/citation_signature","submit_replication":"https://pith.science/pith/JRXAPAQO5WZDCAD2OU2HMQKSQX/action/replication_record"}},"created_at":"2026-07-05T09:00:37.183541+00:00","updated_at":"2026-07-05T09:00:37.183541+00:00"}