{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2019:5YHWOD63SYLQKSRT537EBYXNKM","short_pith_number":"pith:5YHWOD63","schema_version":"1.0","canonical_sha256":"ee0f670fdb9617054a33eefe40e2ed530dc9af8a8f9e799f2248bc614ee2dee9","source":{"kind":"arxiv","id":"1904.01685","version":2},"attestation_state":"computed","paper":{"title":"Measuring Calibration in Deep Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["stat.ML"],"primary_cat":"cs.LG","authors_text":"Dustin Tran, Ghassen Jerfel, Jeremiah Liu, Jeremy Nixon, Linchuan Zhang, Mike Dusenberry, Timothy Nguyen","submitted_at":"2019-04-02T22:10:44Z","abstract_excerpt":"Overconfidence and underconfidence in machine learning classifiers is measured by calibration: the degree to which the probabilities predicted for each class match the accuracy of the classifier on that prediction.\n  How one measures calibration remains a challenge: expected calibration error, the most popular metric, has numerous flaws which we outline, and there is no clear empirical understanding of how its choices affect conclusions in practice, and what recommendations there are to counteract its flaws.\n  In this paper, we perform a comprehensive empirical study of choices in calibration "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"1904.01685","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2019-04-02T22:10:44Z","cross_cats_sorted":["stat.ML"],"title_canon_sha256":"fd5e14da9ae71ac8d21c6c225fe6d0da7c52e6640802dc333461f5a1cafcef49","abstract_canon_sha256":"a5e8f0e66e81a1078fc7cdfe4aed7ffb7fffac69c8216930c6e5ba468fc6a735"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T01:25:34.241256Z","signature_b64":"V66/h8J9PHBvSgFz1yJ8QLJhkbjcdhX+/y7EbM4j2PGmbXhrrXLOE7lONGG6Lazhh38yQtEcBkUCjdN8AViTCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ee0f670fdb9617054a33eefe40e2ed530dc9af8a8f9e799f2248bc614ee2dee9","last_reissued_at":"2026-07-05T01:25:34.240737Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T01:25:34.240737Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Measuring Calibration in Deep Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["stat.ML"],"primary_cat":"cs.LG","authors_text":"Dustin Tran, Ghassen Jerfel, Jeremiah Liu, Jeremy Nixon, Linchuan Zhang, Mike Dusenberry, Timothy Nguyen","submitted_at":"2019-04-02T22:10:44Z","abstract_excerpt":"Overconfidence and underconfidence in machine learning classifiers is measured by calibration: the degree to which the probabilities predicted for each class match the accuracy of the classifier on that prediction.\n  How one measures calibration remains a challenge: expected calibration error, the most popular metric, has numerous flaws which we outline, and there is no clear empirical understanding of how its choices affect conclusions in practice, and what recommendations there are to counteract its flaws.\n  In this paper, we perform a comprehensive empirical study of choices in calibration "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1904.01685","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/1904.01685/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"1904.01685","created_at":"2026-07-05T01:25:34.240800+00:00"},{"alias_kind":"arxiv_version","alias_value":"1904.01685v2","created_at":"2026-07-05T01:25:34.240800+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1904.01685","created_at":"2026-07-05T01:25:34.240800+00:00"},{"alias_kind":"pith_short_12","alias_value":"5YHWOD63SYLQ","created_at":"2026-07-05T01:25:34.240800+00:00"},{"alias_kind":"pith_short_16","alias_value":"5YHWOD63SYLQKSRT","created_at":"2026-07-05T01:25:34.240800+00:00"},{"alias_kind":"pith_short_8","alias_value":"5YHWOD63","created_at":"2026-07-05T01:25:34.240800+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.21116","citing_title":"ConnectomeBench2: A Unified Benchmark for Automated Connectomic Proofreading","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2205.14334","citing_title":"Teaching Models to Express Their Uncertainty in Words","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2604.03203","citing_title":"PR3DICTR: A modular AI framework for medical 3D image-based detection and outcome prediction","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01936","citing_title":"Pandora's Regret: A Proper Scoring Rule for Evaluating Sequential Search","ref_index":107,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/5YHWOD63SYLQKSRT537EBYXNKM","json":"https://pith.science/pith/5YHWOD63SYLQKSRT537EBYXNKM.json","graph_json":"https://pith.science/api/pith-number/5YHWOD63SYLQKSRT537EBYXNKM/graph.json","events_json":"https://pith.science/api/pith-number/5YHWOD63SYLQKSRT537EBYXNKM/events.json","paper":"https://pith.science/paper/5YHWOD63"},"agent_actions":{"view_html":"https://pith.science/pith/5YHWOD63SYLQKSRT537EBYXNKM","download_json":"https://pith.science/pith/5YHWOD63SYLQKSRT537EBYXNKM.json","view_paper":"https://pith.science/paper/5YHWOD63","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=1904.01685&json=true","fetch_graph":"https://pith.science/api/pith-number/5YHWOD63SYLQKSRT537EBYXNKM/graph.json","fetch_events":"https://pith.science/api/pith-number/5YHWOD63SYLQKSRT537EBYXNKM/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/5YHWOD63SYLQKSRT537EBYXNKM/action/timestamp_anchor","attest_storage":"https://pith.science/pith/5YHWOD63SYLQKSRT537EBYXNKM/action/storage_attestation","attest_author":"https://pith.science/pith/5YHWOD63SYLQKSRT537EBYXNKM/action/author_attestation","sign_citation":"https://pith.science/pith/5YHWOD63SYLQKSRT537EBYXNKM/action/citation_signature","submit_replication":"https://pith.science/pith/5YHWOD63SYLQKSRT537EBYXNKM/action/replication_record"}},"created_at":"2026-07-05T01:25:34.240800+00:00","updated_at":"2026-07-05T01:25:34.240800+00:00"}