{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2026:R7CZR3DMOGYOYCXIBMVIGFWBR5","short_pith_number":"pith:R7CZR3DM","schema_version":"1.0","canonical_sha256":"8fc598ec6c71b0ec0ae80b2a8316c18f5fc9aa843790d5d42f8efb214d61f7f9","source":{"kind":"arxiv","id":"2603.16728","version":2},"attestation_state":"computed","paper":{"title":"The Cost of Reasoning: Chain-of-Thought Induces Overconfidence in Vision-Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Emir Konuk, Kevin Smith, Robert Welch","submitted_at":"2026-03-17T16:12:06Z","abstract_excerpt":"Vision-language models (VLMs) are increasingly deployed in high-stakes settings where reliable uncertainty quantification (UQ) is as important as predictive accuracy. Extended reasoning via chain-of-thought (CoT) prompting or reasoning-trained models has become ubiquitous in modern VLM pipelines, yet its effect on UQ reliability remains poorly understood. Our results show that reasoning tends to degrade the quality of many uncertainty estimates, even when it improves task accuracy. We identify implicit answer conditioning as the primary mechanism: as reasoning traces converge on a conclusion b"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2603.16728","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2026-03-17T16:12:06Z","cross_cats_sorted":[],"title_canon_sha256":"9970419a26bc6a23b890287e945b3110ac11f2c71ce605b72fd9b42b1d9390e5","abstract_canon_sha256":"948c2e6921768300efe4304f3a11284f199291763b952e8d3e9413c3228c7932"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-14T01:20:56.478937Z","signature_b64":"vINIiIahNLPRowQdFC9oIFVcjvS4OW7R40HtG0VHOLgY39JLuITr3iIf/8JdPFAalR06+7PtUA8dclJ3mDlvDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"8fc598ec6c71b0ec0ae80b2a8316c18f5fc9aa843790d5d42f8efb214d61f7f9","last_reissued_at":"2026-07-14T01:20:56.478083Z","signature_status":"signed_v1","first_computed_at":"2026-07-14T01:20:56.478083Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"The Cost of Reasoning: Chain-of-Thought Induces Overconfidence in Vision-Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Emir Konuk, Kevin Smith, Robert Welch","submitted_at":"2026-03-17T16:12:06Z","abstract_excerpt":"Vision-language models (VLMs) are increasingly deployed in high-stakes settings where reliable uncertainty quantification (UQ) is as important as predictive accuracy. Extended reasoning via chain-of-thought (CoT) prompting or reasoning-trained models has become ubiquitous in modern VLM pipelines, yet its effect on UQ reliability remains poorly understood. Our results show that reasoning tends to degrade the quality of many uncertainty estimates, even when it improves task accuracy. We identify implicit answer conditioning as the primary mechanism: as reasoning traces converge on a conclusion b"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2603.16728","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2603.16728/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2603.16728","created_at":"2026-07-14T01:20:56.478472+00:00"},{"alias_kind":"arxiv_version","alias_value":"2603.16728v2","created_at":"2026-07-14T01:20:56.478472+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2603.16728","created_at":"2026-07-14T01:20:56.478472+00:00"},{"alias_kind":"pith_short_12","alias_value":"R7CZR3DMOGYO","created_at":"2026-07-14T01:20:56.478472+00:00"},{"alias_kind":"pith_short_16","alias_value":"R7CZR3DMOGYOYCXI","created_at":"2026-07-14T01:20:56.478472+00:00"},{"alias_kind":"pith_short_8","alias_value":"R7CZR3DM","created_at":"2026-07-14T01:20:56.478472+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":2,"sample":[{"citing_arxiv_id":"2607.08059","citing_title":"When Thinking Hurts: Epistemic Signals in the Reasoning Chains of Visual Language Models","ref_index":9,"is_internal_anchor":true},{"citing_arxiv_id":"2605.19869","citing_title":"Passive Construction Site Safety Monitoring via Persona-Scaffolded Adversarial Chain-of-Thought VLM Verification","ref_index":17,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/R7CZR3DMOGYOYCXIBMVIGFWBR5","json":"https://pith.science/pith/R7CZR3DMOGYOYCXIBMVIGFWBR5.json","graph_json":"https://pith.science/api/pith-number/R7CZR3DMOGYOYCXIBMVIGFWBR5/graph.json","events_json":"https://pith.science/api/pith-number/R7CZR3DMOGYOYCXIBMVIGFWBR5/events.json","paper":"https://pith.science/paper/R7CZR3DM"},"agent_actions":{"view_html":"https://pith.science/pith/R7CZR3DMOGYOYCXIBMVIGFWBR5","download_json":"https://pith.science/pith/R7CZR3DMOGYOYCXIBMVIGFWBR5.json","view_paper":"https://pith.science/paper/R7CZR3DM","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2603.16728&json=true","fetch_graph":"https://pith.science/api/pith-number/R7CZR3DMOGYOYCXIBMVIGFWBR5/graph.json","fetch_events":"https://pith.science/api/pith-number/R7CZR3DMOGYOYCXIBMVIGFWBR5/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/R7CZR3DMOGYOYCXIBMVIGFWBR5/action/timestamp_anchor","attest_storage":"https://pith.science/pith/R7CZR3DMOGYOYCXIBMVIGFWBR5/action/storage_attestation","attest_author":"https://pith.science/pith/R7CZR3DMOGYOYCXIBMVIGFWBR5/action/author_attestation","sign_citation":"https://pith.science/pith/R7CZR3DMOGYOYCXIBMVIGFWBR5/action/citation_signature","submit_replication":"https://pith.science/pith/R7CZR3DMOGYOYCXIBMVIGFWBR5/action/replication_record"}},"created_at":"2026-07-14T01:20:56.478472+00:00","updated_at":"2026-07-14T01:20:56.478472+00:00"}