{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2026:YIFOXS37IXS64WOYQONRNQRSPZ","short_pith_number":"pith:YIFOXS37","schema_version":"1.0","canonical_sha256":"c20aebcb7f45e5ee59d8839b16c2327e7e3176ee13a3a77fa55398c2af8bfbd0","source":{"kind":"arxiv","id":"2608.13430","version":1},"attestation_state":"computed","paper":{"title":"Are You Sure You're Sure? On the Impact of Instruction Tuning on Confidence and Lexical Diversity","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Irina Proskurina, Mayank Kumar, Oyindolapo O. Komolafe","submitted_at":"2026-08-13T16:18:45Z","abstract_excerpt":"Instruction-tuned language models achieve strong performance across a range of generation tasks, but have also recently been shown to exhibit verbalized overconfidence. In question answering, verbalized model overconfidence may be associated with the consistency of the generated supporting rationales. In this paper, we study whether corresponding changes in the lexical diversity of generated answer rationales accompany changes in model confidence induced by instruction tuning. We evaluate three matched base and instruction-tuned models across question-answering benchmarks and find that instruc"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2608.13430","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2026-08-13T16:18:45Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"59bf21589bdf8be240ebd10cf58cb3831782dcee88c51ce5cbcc5ea27910bf9d","abstract_canon_sha256":"a2f57e51de0f2586bbe7dfadeadcbf587b6580d1f681acebd560e894bbe0c90b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-08-14T01:04:25.619084Z","signature_b64":"NuQzjC1UqXIIQS05KWhFGdWtkGAkqONHGpDPqfkZbw/rhDrMgl8AekNVN/hajug/p8l4MgMif7I2EOCCJ3IZCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"c20aebcb7f45e5ee59d8839b16c2327e7e3176ee13a3a77fa55398c2af8bfbd0","last_reissued_at":"2026-08-14T01:04:25.617499Z","signature_status":"signed_v1","first_computed_at":"2026-08-14T01:04:25.617499Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Are You Sure You're Sure? On the Impact of Instruction Tuning on Confidence and Lexical Diversity","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Irina Proskurina, Mayank Kumar, Oyindolapo O. Komolafe","submitted_at":"2026-08-13T16:18:45Z","abstract_excerpt":"Instruction-tuned language models achieve strong performance across a range of generation tasks, but have also recently been shown to exhibit verbalized overconfidence. In question answering, verbalized model overconfidence may be associated with the consistency of the generated supporting rationales. In this paper, we study whether corresponding changes in the lexical diversity of generated answer rationales accompany changes in model confidence induced by instruction tuning. We evaluate three matched base and instruction-tuned models across question-answering benchmarks and find that instruc"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2608.13430","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2608.13430/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2608.13430","created_at":"2026-08-14T01:04:25.618279+00:00"},{"alias_kind":"arxiv_version","alias_value":"2608.13430v1","created_at":"2026-08-14T01:04:25.618279+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2608.13430","created_at":"2026-08-14T01:04:25.618279+00:00"},{"alias_kind":"pith_short_12","alias_value":"YIFOXS37IXS6","created_at":"2026-08-14T01:04:25.618279+00:00"},{"alias_kind":"pith_short_16","alias_value":"YIFOXS37IXS64WOY","created_at":"2026-08-14T01:04:25.618279+00:00"},{"alias_kind":"pith_short_8","alias_value":"YIFOXS37","created_at":"2026-08-14T01:04:25.618279+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/YIFOXS37IXS64WOYQONRNQRSPZ","json":"https://pith.science/pith/YIFOXS37IXS64WOYQONRNQRSPZ.json","graph_json":"https://pith.science/api/pith-number/YIFOXS37IXS64WOYQONRNQRSPZ/graph.json","events_json":"https://pith.science/api/pith-number/YIFOXS37IXS64WOYQONRNQRSPZ/events.json","paper":"https://pith.science/paper/YIFOXS37"},"agent_actions":{"view_html":"https://pith.science/pith/YIFOXS37IXS64WOYQONRNQRSPZ","download_json":"https://pith.science/pith/YIFOXS37IXS64WOYQONRNQRSPZ.json","view_paper":"https://pith.science/paper/YIFOXS37","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2608.13430&json=true","fetch_graph":"https://pith.science/api/pith-number/YIFOXS37IXS64WOYQONRNQRSPZ/graph.json","fetch_events":"https://pith.science/api/pith-number/YIFOXS37IXS64WOYQONRNQRSPZ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/YIFOXS37IXS64WOYQONRNQRSPZ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/YIFOXS37IXS64WOYQONRNQRSPZ/action/storage_attestation","attest_author":"https://pith.science/pith/YIFOXS37IXS64WOYQONRNQRSPZ/action/author_attestation","sign_citation":"https://pith.science/pith/YIFOXS37IXS64WOYQONRNQRSPZ/action/citation_signature","submit_replication":"https://pith.science/pith/YIFOXS37IXS64WOYQONRNQRSPZ/action/replication_record"}},"created_at":"2026-08-14T01:04:25.618279+00:00","updated_at":"2026-08-14T01:04:25.618279+00:00"}