{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:FRVSB3SLE7QE4JOSZSAX73U424","short_pith_number":"pith:FRVSB3SL","schema_version":"1.0","canonical_sha256":"2c6b20ee4b27e04e25d2cc817fee9cd714a5c686629b4a0c85c69b05afb0824d","source":{"kind":"arxiv","id":"2404.12452","version":2},"attestation_state":"computed","paper":{"title":"Characterizing LLM Abstention Behavior in Science QA with Context Perturbations","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Bill Howe, Bingbing Wen, Lucy Lu Wang","submitted_at":"2024-04-18T18:26:43Z","abstract_excerpt":"The correct model response in the face of uncertainty is to abstain from answering a question so as not to mislead the user. In this work, we study the ability of LLMs to abstain from answering context-dependent science questions when provided insufficient or incorrect context. We probe model sensitivity in several settings: removing gold context, replacing gold context with irrelevant context, and providing additional context beyond what is given. In experiments on four QA datasets with six LLMs, we show that performance varies greatly across models, across the type of context provided, and a"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2404.12452","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-04-18T18:26:43Z","cross_cats_sorted":[],"title_canon_sha256":"8735366f025470933b1d6dfc22007aaa0cdaf86faaba6e7ab22a94a5b57e7bd7","abstract_canon_sha256":"8fb2ff8e069610af03093c1ac8d843fa87406d7d91522b1b6d491371f40f1490"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:16:14.525336Z","signature_b64":"2snOmQ3X39cqws8kCw2Fkd/t9zLy0vW4Ub6OLW16hk+lJemgwD2rVw7AR4dcH0pS/00KX8FSQdLmQXg9OHTjAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"2c6b20ee4b27e04e25d2cc817fee9cd714a5c686629b4a0c85c69b05afb0824d","last_reissued_at":"2026-07-05T09:16:14.524926Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:16:14.524926Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Characterizing LLM Abstention Behavior in Science QA with Context Perturbations","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Bill Howe, Bingbing Wen, Lucy Lu Wang","submitted_at":"2024-04-18T18:26:43Z","abstract_excerpt":"The correct model response in the face of uncertainty is to abstain from answering a question so as not to mislead the user. In this work, we study the ability of LLMs to abstain from answering context-dependent science questions when provided insufficient or incorrect context. We probe model sensitivity in several settings: removing gold context, replacing gold context with irrelevant context, and providing additional context beyond what is given. In experiments on four QA datasets with six LLMs, we show that performance varies greatly across models, across the type of context provided, and a"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2404.12452","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2404.12452/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2404.12452","created_at":"2026-07-05T09:16:14.524983+00:00"},{"alias_kind":"arxiv_version","alias_value":"2404.12452v2","created_at":"2026-07-05T09:16:14.524983+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2404.12452","created_at":"2026-07-05T09:16:14.524983+00:00"},{"alias_kind":"pith_short_12","alias_value":"FRVSB3SLE7QE","created_at":"2026-07-05T09:16:14.524983+00:00"},{"alias_kind":"pith_short_16","alias_value":"FRVSB3SLE7QE4JOS","created_at":"2026-07-05T09:16:14.524983+00:00"},{"alias_kind":"pith_short_8","alias_value":"FRVSB3SL","created_at":"2026-07-05T09:16:14.524983+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2603.22161","citing_title":"Causal Evidence that Language Models use Confidence to Drive Behavior","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2506.10060","citing_title":"Textual Bayes: Quantifying Prompt Uncertainty in LLM-Based Systems","ref_index":67,"is_internal_anchor":false},{"citing_arxiv_id":"2604.20854","citing_title":"ERA: Evidence-based Reliability Alignment for Honest Retrieval-Augmented Generation","ref_index":40,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/FRVSB3SLE7QE4JOSZSAX73U424","json":"https://pith.science/pith/FRVSB3SLE7QE4JOSZSAX73U424.json","graph_json":"https://pith.science/api/pith-number/FRVSB3SLE7QE4JOSZSAX73U424/graph.json","events_json":"https://pith.science/api/pith-number/FRVSB3SLE7QE4JOSZSAX73U424/events.json","paper":"https://pith.science/paper/FRVSB3SL"},"agent_actions":{"view_html":"https://pith.science/pith/FRVSB3SLE7QE4JOSZSAX73U424","download_json":"https://pith.science/pith/FRVSB3SLE7QE4JOSZSAX73U424.json","view_paper":"https://pith.science/paper/FRVSB3SL","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2404.12452&json=true","fetch_graph":"https://pith.science/api/pith-number/FRVSB3SLE7QE4JOSZSAX73U424/graph.json","fetch_events":"https://pith.science/api/pith-number/FRVSB3SLE7QE4JOSZSAX73U424/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/FRVSB3SLE7QE4JOSZSAX73U424/action/timestamp_anchor","attest_storage":"https://pith.science/pith/FRVSB3SLE7QE4JOSZSAX73U424/action/storage_attestation","attest_author":"https://pith.science/pith/FRVSB3SLE7QE4JOSZSAX73U424/action/author_attestation","sign_citation":"https://pith.science/pith/FRVSB3SLE7QE4JOSZSAX73U424/action/citation_signature","submit_replication":"https://pith.science/pith/FRVSB3SLE7QE4JOSZSAX73U424/action/replication_record"}},"created_at":"2026-07-05T09:16:14.524983+00:00","updated_at":"2026-07-05T09:16:14.524983+00:00"}