{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:2RJ5LYIOIMHZXPHVXFNAVUZZMH","short_pith_number":"pith:2RJ5LYIO","schema_version":"1.0","canonical_sha256":"d453d5e10e430f9bbcf5b95a0ad33961d5bf6b036b97e6e4aaa7389d76f62b42","source":{"kind":"arxiv","id":"2408.05120","version":1},"attestation_state":"computed","paper":{"title":"Cautious Calibration in Binary Classification","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Joonas J\\\"arve, Mari-Liis Allikivi, Meelis Kull","submitted_at":"2024-08-09T15:19:40Z","abstract_excerpt":"Being cautious is crucial for enhancing the trustworthiness of machine learning systems integrated into decision-making pipelines. Although calibrated probabilities help in optimal decision-making, perfect calibration remains unattainable, leading to estimates that fluctuate between under- and overconfidence. This becomes a critical issue in high-risk scenarios, where even occasional overestimation can lead to extreme expected costs. In these scenarios, it is important for each predicted probability to lean towards underconfidence, rather than just achieving an average balance. In this study, "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2408.05120","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-08-09T15:19:40Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"11e2e3800105503d4c225dd37dabede60e24673835b2097136484a8a3e9bef05","abstract_canon_sha256":"185d0b925d5398fe8edc8df266035beb2c4e818d2aa5382f6bef5cccf56a6a9c"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:53:57.142617Z","signature_b64":"fQU58sz4AuGpiNDkCOX7/tP7ilyox2iYMfvX7+RrO3sqKWlo8CWwMf1SF8z0fR+t+gU0PHTF4iwhhPPkb/HTDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"d453d5e10e430f9bbcf5b95a0ad33961d5bf6b036b97e6e4aaa7389d76f62b42","last_reissued_at":"2026-07-05T08:53:57.142249Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:53:57.142249Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Cautious Calibration in Binary Classification","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Joonas J\\\"arve, Mari-Liis Allikivi, Meelis Kull","submitted_at":"2024-08-09T15:19:40Z","abstract_excerpt":"Being cautious is crucial for enhancing the trustworthiness of machine learning systems integrated into decision-making pipelines. Although calibrated probabilities help in optimal decision-making, perfect calibration remains unattainable, leading to estimates that fluctuate between under- and overconfidence. This becomes a critical issue in high-risk scenarios, where even occasional overestimation can lead to extreme expected costs. In these scenarios, it is important for each predicted probability to lean towards underconfidence, rather than just achieving an average balance. In this study, "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2408.05120","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2408.05120/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2408.05120","created_at":"2026-07-05T08:53:57.142304+00:00"},{"alias_kind":"arxiv_version","alias_value":"2408.05120v1","created_at":"2026-07-05T08:53:57.142304+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2408.05120","created_at":"2026-07-05T08:53:57.142304+00:00"},{"alias_kind":"pith_short_12","alias_value":"2RJ5LYIOIMHZ","created_at":"2026-07-05T08:53:57.142304+00:00"},{"alias_kind":"pith_short_16","alias_value":"2RJ5LYIOIMHZXPHV","created_at":"2026-07-05T08:53:57.142304+00:00"},{"alias_kind":"pith_short_8","alias_value":"2RJ5LYIO","created_at":"2026-07-05T08:53:57.142304+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2505.19073","citing_title":"Towards Harmonized Uncertainty Estimation for Large Language Models","ref_index":1,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/2RJ5LYIOIMHZXPHVXFNAVUZZMH","json":"https://pith.science/pith/2RJ5LYIOIMHZXPHVXFNAVUZZMH.json","graph_json":"https://pith.science/api/pith-number/2RJ5LYIOIMHZXPHVXFNAVUZZMH/graph.json","events_json":"https://pith.science/api/pith-number/2RJ5LYIOIMHZXPHVXFNAVUZZMH/events.json","paper":"https://pith.science/paper/2RJ5LYIO"},"agent_actions":{"view_html":"https://pith.science/pith/2RJ5LYIOIMHZXPHVXFNAVUZZMH","download_json":"https://pith.science/pith/2RJ5LYIOIMHZXPHVXFNAVUZZMH.json","view_paper":"https://pith.science/paper/2RJ5LYIO","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2408.05120&json=true","fetch_graph":"https://pith.science/api/pith-number/2RJ5LYIOIMHZXPHVXFNAVUZZMH/graph.json","fetch_events":"https://pith.science/api/pith-number/2RJ5LYIOIMHZXPHVXFNAVUZZMH/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/2RJ5LYIOIMHZXPHVXFNAVUZZMH/action/timestamp_anchor","attest_storage":"https://pith.science/pith/2RJ5LYIOIMHZXPHVXFNAVUZZMH/action/storage_attestation","attest_author":"https://pith.science/pith/2RJ5LYIOIMHZXPHVXFNAVUZZMH/action/author_attestation","sign_citation":"https://pith.science/pith/2RJ5LYIOIMHZXPHVXFNAVUZZMH/action/citation_signature","submit_replication":"https://pith.science/pith/2RJ5LYIOIMHZXPHVXFNAVUZZMH/action/replication_record"}},"created_at":"2026-07-05T08:53:57.142304+00:00","updated_at":"2026-07-05T08:53:57.142304+00:00"}