{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2026:PUJGITPD3O77H7UKBTSDKAEAB5","short_pith_number":"pith:PUJGITPD","schema_version":"1.0","canonical_sha256":"7d12644de3dbbff3fe8a0ce43500800f7dfbd74b00ce276421d9df0052b3669f","source":{"kind":"arxiv","id":"2608.07827","version":1},"attestation_state":"computed","paper":{"title":"From token probabilities to calibrated confidence: An empirical study of mathematical question answering","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.LG","authors_text":"Avery Ma, Leila Pishdad, Lorne Schell, Vin Bhaskara","submitted_at":"2026-08-08T00:00:19Z","abstract_excerpt":"Confidence estimation for large language models (LLMs) aims to estimate the probability that a generated answer is correct, while calibration aligns these estimates with empirical accuracy. Prior work has shown that token probabilities are often overconfident, we investigate whether these readily available signals can nevertheless provide well-calibrated confidence estimation for mathematical question answering. We compare single-pass estimators, which reuse token probabilities from the original generation, with multi-pass estimators, which obtain additional confidence signals through verifica"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2608.07827","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2026-08-08T00:00:19Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"8ec5665932c28b45a3666224f43055ad55d7bf572c8c43a96e6441e5f78849a3","abstract_canon_sha256":"e4ffe2754a7fcaf8d4c3c9101ca16e13505c788433eee427f9f0239cbf29d941"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-08-11T01:18:33.964523Z","signature_b64":"VIilfKxJoJR5wWnKCZlrVZtmRwg9taEWHPddIQepriFIi6fRk705Xvt7nK4U5TeU2X+oqf/b2fvvVIJ/7SuPBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"7d12644de3dbbff3fe8a0ce43500800f7dfbd74b00ce276421d9df0052b3669f","last_reissued_at":"2026-08-11T01:18:33.962282Z","signature_status":"signed_v1","first_computed_at":"2026-08-11T01:18:33.962282Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"From token probabilities to calibrated confidence: An empirical study of mathematical question answering","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.LG","authors_text":"Avery Ma, Leila Pishdad, Lorne Schell, Vin Bhaskara","submitted_at":"2026-08-08T00:00:19Z","abstract_excerpt":"Confidence estimation for large language models (LLMs) aims to estimate the probability that a generated answer is correct, while calibration aligns these estimates with empirical accuracy. Prior work has shown that token probabilities are often overconfident, we investigate whether these readily available signals can nevertheless provide well-calibrated confidence estimation for mathematical question answering. We compare single-pass estimators, which reuse token probabilities from the original generation, with multi-pass estimators, which obtain additional confidence signals through verifica"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2608.07827","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2608.07827/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2608.07827","created_at":"2026-08-11T01:18:33.962903+00:00"},{"alias_kind":"arxiv_version","alias_value":"2608.07827v1","created_at":"2026-08-11T01:18:33.962903+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2608.07827","created_at":"2026-08-11T01:18:33.962903+00:00"},{"alias_kind":"pith_short_12","alias_value":"PUJGITPD3O77","created_at":"2026-08-11T01:18:33.962903+00:00"},{"alias_kind":"pith_short_16","alias_value":"PUJGITPD3O77H7UK","created_at":"2026-08-11T01:18:33.962903+00:00"},{"alias_kind":"pith_short_8","alias_value":"PUJGITPD","created_at":"2026-08-11T01:18:33.962903+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/PUJGITPD3O77H7UKBTSDKAEAB5","json":"https://pith.science/pith/PUJGITPD3O77H7UKBTSDKAEAB5.json","graph_json":"https://pith.science/api/pith-number/PUJGITPD3O77H7UKBTSDKAEAB5/graph.json","events_json":"https://pith.science/api/pith-number/PUJGITPD3O77H7UKBTSDKAEAB5/events.json","paper":"https://pith.science/paper/PUJGITPD"},"agent_actions":{"view_html":"https://pith.science/pith/PUJGITPD3O77H7UKBTSDKAEAB5","download_json":"https://pith.science/pith/PUJGITPD3O77H7UKBTSDKAEAB5.json","view_paper":"https://pith.science/paper/PUJGITPD","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2608.07827&json=true","fetch_graph":"https://pith.science/api/pith-number/PUJGITPD3O77H7UKBTSDKAEAB5/graph.json","fetch_events":"https://pith.science/api/pith-number/PUJGITPD3O77H7UKBTSDKAEAB5/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/PUJGITPD3O77H7UKBTSDKAEAB5/action/timestamp_anchor","attest_storage":"https://pith.science/pith/PUJGITPD3O77H7UKBTSDKAEAB5/action/storage_attestation","attest_author":"https://pith.science/pith/PUJGITPD3O77H7UKBTSDKAEAB5/action/author_attestation","sign_citation":"https://pith.science/pith/PUJGITPD3O77H7UKBTSDKAEAB5/action/citation_signature","submit_replication":"https://pith.science/pith/PUJGITPD3O77H7UKBTSDKAEAB5/action/replication_record"}},"created_at":"2026-08-11T01:18:33.962903+00:00","updated_at":"2026-08-11T01:18:33.962903+00:00"}