{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:QG3TS4SBE4SZOXHPTBKAOOWHHL","short_pith_number":"pith:QG3TS4SB","schema_version":"1.0","canonical_sha256":"81b73972412725975cef9854073ac73afc077c1a5fa2b4c004e5df9bf4b1bf61","source":{"kind":"arxiv","id":"2106.07998","version":2},"attestation_state":"computed","paper":{"title":"Revisiting the Calibration of Modern Neural Networks","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CV"],"primary_cat":"cs.LG","authors_text":"Dustin Tran, Frances Hubis, Josip Djolonga, Mario Lucic, Matthias Minderer, Neil Houlsby, Rob Romijnders, Xiaohua Zhai","submitted_at":"2021-06-15T09:24:43Z","abstract_excerpt":"Accurate estimation of predictive uncertainty (model calibration) is essential for the safe application of neural networks. Many instances of miscalibration in modern neural networks have been reported, suggesting a trend that newer, more accurate models produce poorly calibrated predictions. Here, we revisit this question for recent state-of-the-art image classification models. We systematically relate model calibration and accuracy, and find that the most recent models, notably those not using convolutions, are among the best calibrated. Trends observed in prior model generations, such as de"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2106.07998","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2021-06-15T09:24:43Z","cross_cats_sorted":["cs.CV"],"title_canon_sha256":"843e087ac9042d19fa1f570f8ff8170d04396f33153af26c23510ca86cc53b07","abstract_canon_sha256":"4167c0a35f6bd617d968603ff3dbd381d6d5a9c4f00e7f7663cc2e49ddf79b4b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T03:25:32.206755Z","signature_b64":"APkun5fljoEMbyUgJyPtaSY9dqsnBipxChwAYEUMcjW1G+HZXZMyO6TrhkJMF1vmvLIA+bB9VX1kmFxljRaeAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"81b73972412725975cef9854073ac73afc077c1a5fa2b4c004e5df9bf4b1bf61","last_reissued_at":"2026-07-05T03:25:32.206328Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T03:25:32.206328Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Revisiting the Calibration of Modern Neural Networks","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CV"],"primary_cat":"cs.LG","authors_text":"Dustin Tran, Frances Hubis, Josip Djolonga, Mario Lucic, Matthias Minderer, Neil Houlsby, Rob Romijnders, Xiaohua Zhai","submitted_at":"2021-06-15T09:24:43Z","abstract_excerpt":"Accurate estimation of predictive uncertainty (model calibration) is essential for the safe application of neural networks. Many instances of miscalibration in modern neural networks have been reported, suggesting a trend that newer, more accurate models produce poorly calibrated predictions. Here, we revisit this question for recent state-of-the-art image classification models. We systematically relate model calibration and accuracy, and find that the most recent models, notably those not using convolutions, are among the best calibrated. Trends observed in prior model generations, such as de"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2106.07998","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2106.07998/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2106.07998","created_at":"2026-07-05T03:25:32.206391+00:00"},{"alias_kind":"arxiv_version","alias_value":"2106.07998v2","created_at":"2026-07-05T03:25:32.206391+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2106.07998","created_at":"2026-07-05T03:25:32.206391+00:00"},{"alias_kind":"pith_short_12","alias_value":"QG3TS4SBE4SZ","created_at":"2026-07-05T03:25:32.206391+00:00"},{"alias_kind":"pith_short_16","alias_value":"QG3TS4SBE4SZOXHP","created_at":"2026-07-05T03:25:32.206391+00:00"},{"alias_kind":"pith_short_8","alias_value":"QG3TS4SB","created_at":"2026-07-05T03:25:32.206391+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.09881","citing_title":"Toward Calibrated, Fair, and accurate Deepfake Detection","ref_index":287,"is_internal_anchor":false},{"citing_arxiv_id":"2605.00195","citing_title":"Diversity in Large Language Models under Supervised Fine-Tuning","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2604.25154","citing_title":"Prior-Aligned Data Cleaning for Tabular Foundation Models","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2604.19444","citing_title":"Unsupervised Confidence Calibration for Reasoning LLMs from a Single Generation","ref_index":181,"is_internal_anchor":false},{"citing_arxiv_id":"2604.19444","citing_title":"Unsupervised Confidence Calibration for Reasoning LLMs from a Single Generation","ref_index":62,"is_internal_anchor":false},{"citing_arxiv_id":"2605.00195","citing_title":"Diversity in Large Language Models under Supervised Fine-Tuning","ref_index":3,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/QG3TS4SBE4SZOXHPTBKAOOWHHL","json":"https://pith.science/pith/QG3TS4SBE4SZOXHPTBKAOOWHHL.json","graph_json":"https://pith.science/api/pith-number/QG3TS4SBE4SZOXHPTBKAOOWHHL/graph.json","events_json":"https://pith.science/api/pith-number/QG3TS4SBE4SZOXHPTBKAOOWHHL/events.json","paper":"https://pith.science/paper/QG3TS4SB"},"agent_actions":{"view_html":"https://pith.science/pith/QG3TS4SBE4SZOXHPTBKAOOWHHL","download_json":"https://pith.science/pith/QG3TS4SBE4SZOXHPTBKAOOWHHL.json","view_paper":"https://pith.science/paper/QG3TS4SB","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2106.07998&json=true","fetch_graph":"https://pith.science/api/pith-number/QG3TS4SBE4SZOXHPTBKAOOWHHL/graph.json","fetch_events":"https://pith.science/api/pith-number/QG3TS4SBE4SZOXHPTBKAOOWHHL/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/QG3TS4SBE4SZOXHPTBKAOOWHHL/action/timestamp_anchor","attest_storage":"https://pith.science/pith/QG3TS4SBE4SZOXHPTBKAOOWHHL/action/storage_attestation","attest_author":"https://pith.science/pith/QG3TS4SBE4SZOXHPTBKAOOWHHL/action/author_attestation","sign_citation":"https://pith.science/pith/QG3TS4SBE4SZOXHPTBKAOOWHHL/action/citation_signature","submit_replication":"https://pith.science/pith/QG3TS4SBE4SZOXHPTBKAOOWHHL/action/replication_record"}},"created_at":"2026-07-05T03:25:32.206391+00:00","updated_at":"2026-07-05T03:25:32.206391+00:00"}