{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:Z5W7HJVCQ6P4NMZZYCYRC3GLEX","short_pith_number":"pith:Z5W7HJVC","schema_version":"1.0","canonical_sha256":"cf6df3a6a2879fc6b339c0b1116ccb25c6da5acf78494402db9d726c33822e16","source":{"kind":"arxiv","id":"2407.18418","version":3},"attestation_state":"computed","paper":{"title":"Know Your Limits: A Survey of Abstention in Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Bill Howe, Bingbing Wen, Chenjun Xu, Jihan Yao, Lucy Lu Wang, Shangbin Feng, Yulia Tsvetkov","submitted_at":"2024-07-25T22:31:50Z","abstract_excerpt":"Abstention, the refusal of large language models (LLMs) to provide an answer, is increasingly recognized for its potential to mitigate hallucinations and enhance safety in LLM systems. In this survey, we introduce a framework to examine abstention from three perspectives: the query, the model, and human values. We organize the literature on abstention methods, benchmarks, and evaluation metrics using this framework, and discuss merits and limitations of prior work. We further identify and motivate areas for future research, such as whether abstention can be achieved as a meta-capability that t"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2407.18418","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-07-25T22:31:50Z","cross_cats_sorted":[],"title_canon_sha256":"f761f79954ff461726c640aeae63d86dd2b124082d68a5e3ca51f8a43362bf28","abstract_canon_sha256":"81e0f121646ac85c979c1e54ed5d3224feae47d09a9365aa9c66c8d24a228faf"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:13:02.136064Z","signature_b64":"tHVS9QvF9sBwXt4j8YZXmr4r0PrJTvs8tcfSlAZ4fKWc5B/KzG6B8rR/dudzaM0J3GWZHemINxW7+3hFGbQhDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"cf6df3a6a2879fc6b339c0b1116ccb25c6da5acf78494402db9d726c33822e16","last_reissued_at":"2026-07-05T10:13:02.135582Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:13:02.135582Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Know Your Limits: A Survey of Abstention in Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Bill Howe, Bingbing Wen, Chenjun Xu, Jihan Yao, Lucy Lu Wang, Shangbin Feng, Yulia Tsvetkov","submitted_at":"2024-07-25T22:31:50Z","abstract_excerpt":"Abstention, the refusal of large language models (LLMs) to provide an answer, is increasingly recognized for its potential to mitigate hallucinations and enhance safety in LLM systems. In this survey, we introduce a framework to examine abstention from three perspectives: the query, the model, and human values. We organize the literature on abstention methods, benchmarks, and evaluation metrics using this framework, and discuss merits and limitations of prior work. We further identify and motivate areas for future research, such as whether abstention can be achieved as a meta-capability that t"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2407.18418","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2407.18418/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2407.18418","created_at":"2026-07-05T10:13:02.135642+00:00"},{"alias_kind":"arxiv_version","alias_value":"2407.18418v3","created_at":"2026-07-05T10:13:02.135642+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2407.18418","created_at":"2026-07-05T10:13:02.135642+00:00"},{"alias_kind":"pith_short_12","alias_value":"Z5W7HJVCQ6P4","created_at":"2026-07-05T10:13:02.135642+00:00"},{"alias_kind":"pith_short_16","alias_value":"Z5W7HJVCQ6P4NMZZ","created_at":"2026-07-05T10:13:02.135642+00:00"},{"alias_kind":"pith_short_8","alias_value":"Z5W7HJVC","created_at":"2026-07-05T10:13:02.135642+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":7,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.24370","citing_title":"When Helpfulness Overrides Causal Caution: Context-Dependent Suppression and Recovery in LLMs","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2606.19950","citing_title":"Confidence Calibration for Multimodal LLMs: An Empirical Study through Medical VQA","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2606.18668","citing_title":"EARS: Explanatory Abstention for Reliable Sub-Agent Modeling in Large-scale Multi-Agent Systems","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2605.14744","citing_title":"Mechanical Enforcement for LLM Governance:Evidence of Governance-Task Decoupling in Financial Decision Systems","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07919","citing_title":"MedVIGIL: Evaluating Trustworthy Medical VLMs Under Broken Visual Evidence","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18792","citing_title":"Trust or Abstain? A Self-Aware RAG Approach","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07919","citing_title":"MedVIGIL: Evaluating Trustworthy Medical VLMs Under Broken Visual Evidence","ref_index":29,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/Z5W7HJVCQ6P4NMZZYCYRC3GLEX","json":"https://pith.science/pith/Z5W7HJVCQ6P4NMZZYCYRC3GLEX.json","graph_json":"https://pith.science/api/pith-number/Z5W7HJVCQ6P4NMZZYCYRC3GLEX/graph.json","events_json":"https://pith.science/api/pith-number/Z5W7HJVCQ6P4NMZZYCYRC3GLEX/events.json","paper":"https://pith.science/paper/Z5W7HJVC"},"agent_actions":{"view_html":"https://pith.science/pith/Z5W7HJVCQ6P4NMZZYCYRC3GLEX","download_json":"https://pith.science/pith/Z5W7HJVCQ6P4NMZZYCYRC3GLEX.json","view_paper":"https://pith.science/paper/Z5W7HJVC","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2407.18418&json=true","fetch_graph":"https://pith.science/api/pith-number/Z5W7HJVCQ6P4NMZZYCYRC3GLEX/graph.json","fetch_events":"https://pith.science/api/pith-number/Z5W7HJVCQ6P4NMZZYCYRC3GLEX/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/Z5W7HJVCQ6P4NMZZYCYRC3GLEX/action/timestamp_anchor","attest_storage":"https://pith.science/pith/Z5W7HJVCQ6P4NMZZYCYRC3GLEX/action/storage_attestation","attest_author":"https://pith.science/pith/Z5W7HJVCQ6P4NMZZYCYRC3GLEX/action/author_attestation","sign_citation":"https://pith.science/pith/Z5W7HJVCQ6P4NMZZYCYRC3GLEX/action/citation_signature","submit_replication":"https://pith.science/pith/Z5W7HJVCQ6P4NMZZYCYRC3GLEX/action/replication_record"}},"created_at":"2026-07-05T10:13:02.135642+00:00","updated_at":"2026-07-05T10:13:02.135642+00:00"}