{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:W4NNGH3BGGLGMQBMSHONPDEJLB","short_pith_number":"pith:W4NNGH3B","schema_version":"1.0","canonical_sha256":"b71ad31f61319666402c91dcd78c89585420110893c5304834abe2de4378494f","source":{"kind":"arxiv","id":"2402.08733","version":2},"attestation_state":"computed","paper":{"title":"Experts Don't Cheat: Learning What You Don't Know By Predicting Pairs","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Chris J. Maddison, Daniel D. Johnson, Daniel Tarlow, David Duvenaud","submitted_at":"2024-02-13T19:01:45Z","abstract_excerpt":"Identifying how much a model ${\\widehat{p}}_{\\theta}(Y|X)$ knows about the stochastic real-world process $p(Y|X)$ it was trained on is important to ensure it avoids producing incorrect or \"hallucinated\" answers or taking unsafe actions. But this is difficult for generative models because probabilistic predictions do not distinguish between per-response noise (aleatoric uncertainty) and lack of knowledge about the process (epistemic uncertainty), and existing epistemic uncertainty quantification techniques tend to be overconfident when the model underfits. We propose a general strategy for teac"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2402.08733","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-02-13T19:01:45Z","cross_cats_sorted":[],"title_canon_sha256":"a707ded989e7a8c2fa9bf5707925d0e167dfeb5e4d5104123b56ca9d69519068","abstract_canon_sha256":"3796f48416faa9985343c960f44e0733108aee031c35bc1701edaba6bc1cfd1e"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:24:01.547403Z","signature_b64":"fLZyT+GHxapPKMC9/pQ7KRlYoYf9glM4k8E2mPS1HO3Bcnl/f9PLS/VJJpSPjWImz8B6aFj2G++sfC+WNMQjCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b71ad31f61319666402c91dcd78c89585420110893c5304834abe2de4378494f","last_reissued_at":"2026-07-05T08:24:01.546868Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:24:01.546868Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Experts Don't Cheat: Learning What You Don't Know By Predicting Pairs","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Chris J. Maddison, Daniel D. Johnson, Daniel Tarlow, David Duvenaud","submitted_at":"2024-02-13T19:01:45Z","abstract_excerpt":"Identifying how much a model ${\\widehat{p}}_{\\theta}(Y|X)$ knows about the stochastic real-world process $p(Y|X)$ it was trained on is important to ensure it avoids producing incorrect or \"hallucinated\" answers or taking unsafe actions. But this is difficult for generative models because probabilistic predictions do not distinguish between per-response noise (aleatoric uncertainty) and lack of knowledge about the process (epistemic uncertainty), and existing epistemic uncertainty quantification techniques tend to be overconfident when the model underfits. We propose a general strategy for teac"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2402.08733","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2402.08733/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2402.08733","created_at":"2026-07-05T08:24:01.546930+00:00"},{"alias_kind":"arxiv_version","alias_value":"2402.08733v2","created_at":"2026-07-05T08:24:01.546930+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2402.08733","created_at":"2026-07-05T08:24:01.546930+00:00"},{"alias_kind":"pith_short_12","alias_value":"W4NNGH3BGGLG","created_at":"2026-07-05T08:24:01.546930+00:00"},{"alias_kind":"pith_short_16","alias_value":"W4NNGH3BGGLGMQBM","created_at":"2026-07-05T08:24:01.546930+00:00"},{"alias_kind":"pith_short_8","alias_value":"W4NNGH3B","created_at":"2026-07-05T08:24:01.546930+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2604.17200","citing_title":"Calibrating Model-Based Evaluation Metrics for Summarization","ref_index":69,"is_internal_anchor":false},{"citing_arxiv_id":"2604.17112","citing_title":"Complementing Self-Consistency with Cross-Model Disagreement for Uncertainty Quantification","ref_index":15,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/W4NNGH3BGGLGMQBMSHONPDEJLB","json":"https://pith.science/pith/W4NNGH3BGGLGMQBMSHONPDEJLB.json","graph_json":"https://pith.science/api/pith-number/W4NNGH3BGGLGMQBMSHONPDEJLB/graph.json","events_json":"https://pith.science/api/pith-number/W4NNGH3BGGLGMQBMSHONPDEJLB/events.json","paper":"https://pith.science/paper/W4NNGH3B"},"agent_actions":{"view_html":"https://pith.science/pith/W4NNGH3BGGLGMQBMSHONPDEJLB","download_json":"https://pith.science/pith/W4NNGH3BGGLGMQBMSHONPDEJLB.json","view_paper":"https://pith.science/paper/W4NNGH3B","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2402.08733&json=true","fetch_graph":"https://pith.science/api/pith-number/W4NNGH3BGGLGMQBMSHONPDEJLB/graph.json","fetch_events":"https://pith.science/api/pith-number/W4NNGH3BGGLGMQBMSHONPDEJLB/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/W4NNGH3BGGLGMQBMSHONPDEJLB/action/timestamp_anchor","attest_storage":"https://pith.science/pith/W4NNGH3BGGLGMQBMSHONPDEJLB/action/storage_attestation","attest_author":"https://pith.science/pith/W4NNGH3BGGLGMQBMSHONPDEJLB/action/author_attestation","sign_citation":"https://pith.science/pith/W4NNGH3BGGLGMQBMSHONPDEJLB/action/citation_signature","submit_replication":"https://pith.science/pith/W4NNGH3BGGLGMQBMSHONPDEJLB/action/replication_record"}},"created_at":"2026-07-05T08:24:01.546930+00:00","updated_at":"2026-07-05T08:24:01.546930+00:00"}