{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:G4KOUAHR4STOY335G6QDNMZTQO","short_pith_number":"pith:G4KOUAHR","schema_version":"1.0","canonical_sha256":"3714ea00f1e4a6ec6f7d37a036b33383976253d44114b19d5d902bf272d816cc","source":{"kind":"arxiv","id":"2503.17882","version":1},"attestation_state":"computed","paper":{"title":"Think Before Refusal : Triggering Safety Reflection in LLMs to Mitigate False Refusal Behavior","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Barbara Plank, Guangyao Zhai, Nassir Navab, Shengyun Si, Xinpeng Wang","submitted_at":"2025-03-22T23:35:49Z","abstract_excerpt":"Recent advancements in large language models (LLMs) have demonstrated that fine-tuning and human alignment can render LLMs harmless. In practice, such \"harmlessness\" behavior is mainly achieved by training models to reject harmful requests, such as \"Explain how to burn down my neighbor's house\", where the model appropriately declines to respond. However, this approach can inadvertently result in false refusal, where models reject benign queries as well, such as \"Tell me how to kill a Python process\". In this work, we demonstrate that prompting safety reflection before generating a response can"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2503.17882","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2025-03-22T23:35:49Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"61904158185b99d6ca7def1fb589b43278aa38ee2189cbdc83edac184290b44d","abstract_canon_sha256":"cd8da13db9e26ea874ea5f59d01f6207eb964b3571b8e62dda09b077e3ae6ad9"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:37:57.771020Z","signature_b64":"fLv8AQQ5igUQby44vnkiIS6IeXugFlS6h9PCLK9IYaR4EU+N+qzYiLtCebS6hm3FxKJ1Ji6bR065Il9smwglCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"3714ea00f1e4a6ec6f7d37a036b33383976253d44114b19d5d902bf272d816cc","last_reissued_at":"2026-07-05T10:37:57.770520Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:37:57.770520Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Think Before Refusal : Triggering Safety Reflection in LLMs to Mitigate False Refusal Behavior","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Barbara Plank, Guangyao Zhai, Nassir Navab, Shengyun Si, Xinpeng Wang","submitted_at":"2025-03-22T23:35:49Z","abstract_excerpt":"Recent advancements in large language models (LLMs) have demonstrated that fine-tuning and human alignment can render LLMs harmless. In practice, such \"harmlessness\" behavior is mainly achieved by training models to reject harmful requests, such as \"Explain how to burn down my neighbor's house\", where the model appropriately declines to respond. However, this approach can inadvertently result in false refusal, where models reject benign queries as well, such as \"Tell me how to kill a Python process\". In this work, we demonstrate that prompting safety reflection before generating a response can"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2503.17882","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2503.17882/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2503.17882","created_at":"2026-07-05T10:37:57.770581+00:00"},{"alias_kind":"arxiv_version","alias_value":"2503.17882v1","created_at":"2026-07-05T10:37:57.770581+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2503.17882","created_at":"2026-07-05T10:37:57.770581+00:00"},{"alias_kind":"pith_short_12","alias_value":"G4KOUAHR4STO","created_at":"2026-07-05T10:37:57.770581+00:00"},{"alias_kind":"pith_short_16","alias_value":"G4KOUAHR4STOY335","created_at":"2026-07-05T10:37:57.770581+00:00"},{"alias_kind":"pith_short_8","alias_value":"G4KOUAHR","created_at":"2026-07-05T10:37:57.770581+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.31748","citing_title":"Addressing Over-Refusal in LLMs with Competing Rewards","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2606.03601","citing_title":"DDOR: Delta Debugging for Explainable Overrefusal Testing and Repair","ref_index":26,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/G4KOUAHR4STOY335G6QDNMZTQO","json":"https://pith.science/pith/G4KOUAHR4STOY335G6QDNMZTQO.json","graph_json":"https://pith.science/api/pith-number/G4KOUAHR4STOY335G6QDNMZTQO/graph.json","events_json":"https://pith.science/api/pith-number/G4KOUAHR4STOY335G6QDNMZTQO/events.json","paper":"https://pith.science/paper/G4KOUAHR"},"agent_actions":{"view_html":"https://pith.science/pith/G4KOUAHR4STOY335G6QDNMZTQO","download_json":"https://pith.science/pith/G4KOUAHR4STOY335G6QDNMZTQO.json","view_paper":"https://pith.science/paper/G4KOUAHR","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2503.17882&json=true","fetch_graph":"https://pith.science/api/pith-number/G4KOUAHR4STOY335G6QDNMZTQO/graph.json","fetch_events":"https://pith.science/api/pith-number/G4KOUAHR4STOY335G6QDNMZTQO/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/G4KOUAHR4STOY335G6QDNMZTQO/action/timestamp_anchor","attest_storage":"https://pith.science/pith/G4KOUAHR4STOY335G6QDNMZTQO/action/storage_attestation","attest_author":"https://pith.science/pith/G4KOUAHR4STOY335G6QDNMZTQO/action/author_attestation","sign_citation":"https://pith.science/pith/G4KOUAHR4STOY335G6QDNMZTQO/action/citation_signature","submit_replication":"https://pith.science/pith/G4KOUAHR4STOY335G6QDNMZTQO/action/replication_record"}},"created_at":"2026-07-05T10:37:57.770581+00:00","updated_at":"2026-07-05T10:37:57.770581+00:00"}