{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:FVL3JONXAPIWDURWJLCAXVPNP7","short_pith_number":"pith:FVL3JONX","schema_version":"1.0","canonical_sha256":"2d57b4b9b703d161d2364ac40bd5ed7fdb2996e4a7507c4a3498a9bf00f9f23b","source":{"kind":"arxiv","id":"2506.09593","version":1},"attestation_state":"computed","paper":{"title":"Beyond Overconfidence: Foundation Models Redefine Calibration in Deep Neural Networks","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Achim Hekler, Florian Buettner, Lukas Kuhn","submitted_at":"2025-06-11T10:48:36Z","abstract_excerpt":"Reliable uncertainty calibration is essential for safely deploying deep neural networks in high-stakes applications. Deep neural networks are known to exhibit systematic overconfidence, especially under distribution shifts. Although foundation models such as ConvNeXt, EVA and BEiT have demonstrated significant improvements in predictive performance, their calibration properties remain underexplored. This paper presents a comprehensive investigation into the calibration behavior of foundation models, revealing insights that challenge established paradigms. Our empirical analysis shows that thes"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2506.09593","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-06-11T10:48:36Z","cross_cats_sorted":[],"title_canon_sha256":"6d9ca180059f8fcf0440a7f2ce370a045a3a30d8ac37a89835a0616085387c6b","abstract_canon_sha256":"3f5c25c80b60affe5140fc32a59631ac6491fb362ce244f7cfb95aa1cc01ac3d"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:19:39.467940Z","signature_b64":"GyA0O5+BQ3BJ36+c/RDCMb3fo5NpHDYlqRmQstW0b+cWENwptCkWbCr8ZQywkQWG3WxSvRalNcCy94Xj4RFHDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"2d57b4b9b703d161d2364ac40bd5ed7fdb2996e4a7507c4a3498a9bf00f9f23b","last_reissued_at":"2026-07-05T11:19:39.467591Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:19:39.467591Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Beyond Overconfidence: Foundation Models Redefine Calibration in Deep Neural Networks","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Achim Hekler, Florian Buettner, Lukas Kuhn","submitted_at":"2025-06-11T10:48:36Z","abstract_excerpt":"Reliable uncertainty calibration is essential for safely deploying deep neural networks in high-stakes applications. Deep neural networks are known to exhibit systematic overconfidence, especially under distribution shifts. Although foundation models such as ConvNeXt, EVA and BEiT have demonstrated significant improvements in predictive performance, their calibration properties remain underexplored. This paper presents a comprehensive investigation into the calibration behavior of foundation models, revealing insights that challenge established paradigms. Our empirical analysis shows that thes"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.09593","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2506.09593/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2506.09593","created_at":"2026-07-05T11:19:39.467648+00:00"},{"alias_kind":"arxiv_version","alias_value":"2506.09593v1","created_at":"2026-07-05T11:19:39.467648+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.09593","created_at":"2026-07-05T11:19:39.467648+00:00"},{"alias_kind":"pith_short_12","alias_value":"FVL3JONXAPIW","created_at":"2026-07-05T11:19:39.467648+00:00"},{"alias_kind":"pith_short_16","alias_value":"FVL3JONXAPIWDURW","created_at":"2026-07-05T11:19:39.467648+00:00"},{"alias_kind":"pith_short_8","alias_value":"FVL3JONX","created_at":"2026-07-05T11:19:39.467648+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.30188","citing_title":"CalArena: A Large-Scale Post-Hoc Calibration Benchmark","ref_index":31,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/FVL3JONXAPIWDURWJLCAXVPNP7","json":"https://pith.science/pith/FVL3JONXAPIWDURWJLCAXVPNP7.json","graph_json":"https://pith.science/api/pith-number/FVL3JONXAPIWDURWJLCAXVPNP7/graph.json","events_json":"https://pith.science/api/pith-number/FVL3JONXAPIWDURWJLCAXVPNP7/events.json","paper":"https://pith.science/paper/FVL3JONX"},"agent_actions":{"view_html":"https://pith.science/pith/FVL3JONXAPIWDURWJLCAXVPNP7","download_json":"https://pith.science/pith/FVL3JONXAPIWDURWJLCAXVPNP7.json","view_paper":"https://pith.science/paper/FVL3JONX","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2506.09593&json=true","fetch_graph":"https://pith.science/api/pith-number/FVL3JONXAPIWDURWJLCAXVPNP7/graph.json","fetch_events":"https://pith.science/api/pith-number/FVL3JONXAPIWDURWJLCAXVPNP7/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/FVL3JONXAPIWDURWJLCAXVPNP7/action/timestamp_anchor","attest_storage":"https://pith.science/pith/FVL3JONXAPIWDURWJLCAXVPNP7/action/storage_attestation","attest_author":"https://pith.science/pith/FVL3JONXAPIWDURWJLCAXVPNP7/action/author_attestation","sign_citation":"https://pith.science/pith/FVL3JONXAPIWDURWJLCAXVPNP7/action/citation_signature","submit_replication":"https://pith.science/pith/FVL3JONXAPIWDURWJLCAXVPNP7/action/replication_record"}},"created_at":"2026-07-05T11:19:39.467648+00:00","updated_at":"2026-07-05T11:19:39.467648+00:00"}