{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2026:IWMQZVRNT5YYD5YVA2O4YPKHSA","short_pith_number":"pith:IWMQZVRN","schema_version":"1.0","canonical_sha256":"45990cd62d9f7181f715069dcc3d47902ab6b9959643fa93ed2fc3098d97b6ba","source":{"kind":"arxiv","id":"2608.12008","version":1},"attestation_state":"computed","paper":{"title":"Asymptotic Risk Calibration for Selective Question Answering","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Shufan Lin, Sijin Dong","submitted_at":"2026-08-12T12:45:45Z","abstract_excerpt":"Large language models (LLMs) may generate fluent but incorrect answers, making uncertainty quantification important for reliable question answering. However, heuristic uncertainty scores cannot perfectly distinguish correct predictions from incorrect ones, and directly applying a fixed uncertainty threshold provides no statistical control over the error rate among accepted answers. To address this limitation, we propose A-CRC-QA, a post-hoc calibration framework for uncertainty-aware selective question answering. The proposed method reformulates selection-conditioned error control as a linear "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2608.12008","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2026-08-12T12:45:45Z","cross_cats_sorted":[],"title_canon_sha256":"5a8a3491399c7dc05c3e8f8e509a40c4c2b7c3e6e1c0a04c6f0c3775ce686288","abstract_canon_sha256":"0be1a755adb66985edcc5145b4a9f851bf4c87f591d90caee5fbea90b7520cbe"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-08-13T01:29:14.685435Z","signature_b64":"QDUxcD8KXKa7IlXedc2RJtBTbRBHE/PblU4Bls+79nSgTfNcbI08akVWYnL/xq6qw6JhJUMaThx5ydAQnQfjBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"45990cd62d9f7181f715069dcc3d47902ab6b9959643fa93ed2fc3098d97b6ba","last_reissued_at":"2026-08-13T01:29:14.683230Z","signature_status":"signed_v1","first_computed_at":"2026-08-13T01:29:14.683230Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Asymptotic Risk Calibration for Selective Question Answering","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Shufan Lin, Sijin Dong","submitted_at":"2026-08-12T12:45:45Z","abstract_excerpt":"Large language models (LLMs) may generate fluent but incorrect answers, making uncertainty quantification important for reliable question answering. However, heuristic uncertainty scores cannot perfectly distinguish correct predictions from incorrect ones, and directly applying a fixed uncertainty threshold provides no statistical control over the error rate among accepted answers. To address this limitation, we propose A-CRC-QA, a post-hoc calibration framework for uncertainty-aware selective question answering. The proposed method reformulates selection-conditioned error control as a linear "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2608.12008","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2608.12008/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2608.12008","created_at":"2026-08-13T01:29:14.684333+00:00"},{"alias_kind":"arxiv_version","alias_value":"2608.12008v1","created_at":"2026-08-13T01:29:14.684333+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2608.12008","created_at":"2026-08-13T01:29:14.684333+00:00"},{"alias_kind":"pith_short_12","alias_value":"IWMQZVRNT5YY","created_at":"2026-08-13T01:29:14.684333+00:00"},{"alias_kind":"pith_short_16","alias_value":"IWMQZVRNT5YYD5YV","created_at":"2026-08-13T01:29:14.684333+00:00"},{"alias_kind":"pith_short_8","alias_value":"IWMQZVRN","created_at":"2026-08-13T01:29:14.684333+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/IWMQZVRNT5YYD5YVA2O4YPKHSA","json":"https://pith.science/pith/IWMQZVRNT5YYD5YVA2O4YPKHSA.json","graph_json":"https://pith.science/api/pith-number/IWMQZVRNT5YYD5YVA2O4YPKHSA/graph.json","events_json":"https://pith.science/api/pith-number/IWMQZVRNT5YYD5YVA2O4YPKHSA/events.json","paper":"https://pith.science/paper/IWMQZVRN"},"agent_actions":{"view_html":"https://pith.science/pith/IWMQZVRNT5YYD5YVA2O4YPKHSA","download_json":"https://pith.science/pith/IWMQZVRNT5YYD5YVA2O4YPKHSA.json","view_paper":"https://pith.science/paper/IWMQZVRN","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2608.12008&json=true","fetch_graph":"https://pith.science/api/pith-number/IWMQZVRNT5YYD5YVA2O4YPKHSA/graph.json","fetch_events":"https://pith.science/api/pith-number/IWMQZVRNT5YYD5YVA2O4YPKHSA/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/IWMQZVRNT5YYD5YVA2O4YPKHSA/action/timestamp_anchor","attest_storage":"https://pith.science/pith/IWMQZVRNT5YYD5YVA2O4YPKHSA/action/storage_attestation","attest_author":"https://pith.science/pith/IWMQZVRNT5YYD5YVA2O4YPKHSA/action/author_attestation","sign_citation":"https://pith.science/pith/IWMQZVRNT5YYD5YVA2O4YPKHSA/action/citation_signature","submit_replication":"https://pith.science/pith/IWMQZVRNT5YYD5YVA2O4YPKHSA/action/replication_record"}},"created_at":"2026-08-13T01:29:14.684333+00:00","updated_at":"2026-08-13T01:29:14.684333+00:00"}