{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:DWHCR4ZCW2RNEHGRSWAIULRYM3","short_pith_number":"pith:DWHCR4ZC","schema_version":"1.0","canonical_sha256":"1d8e28f322b6a2d21cd195808a2e3866c3d9574e89b62fc22cf1a7484190054e","source":{"kind":"arxiv","id":"2306.16564","version":4},"attestation_state":"computed","paper":{"title":"Pareto Optimal Learning for Estimating Large Language Model Errors","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["stat.ML"],"primary_cat":"cs.CL","authors_text":"Hoifung Poon, J. Samuel Preston, Mu Wei, Theodore Zhao","submitted_at":"2023-06-28T21:11:15Z","abstract_excerpt":"Large Language Models (LLMs) have shown impressive abilities in many applications. When a concrete and precise answer is desired, it is important to have a quantitative estimation of the potential error rate. However, this can be challenging due to the text-in-text-out nature of generative models. We present a method based on Pareto optimization that generates a risk score to estimate the probability of error in an LLM response by integrating multiple sources of information. We prove theoretically that the error estimator optimized in our framework aligns with the LLM and the information sourc"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2306.16564","kind":"arxiv","version":4},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2023-06-28T21:11:15Z","cross_cats_sorted":["stat.ML"],"title_canon_sha256":"33e6c8dbe71611efcf006a2434ffbc4bab642145257656514d886c9d50fd74cd","abstract_canon_sha256":"5f3d9f13739d0aa97151804ddc164e4adf142d7bc69efff71eb2d2668b5afeae"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:51:28.478968Z","signature_b64":"hKemnpmAW7DSUgDUntAyQmLfN7oeeG16QRdaixsGMbTlWHNhunTJFy7UbAhrAF7G2CJ1b5IdMcT0ZHzkyfKMDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"1d8e28f322b6a2d21cd195808a2e3866c3d9574e89b62fc22cf1a7484190054e","last_reissued_at":"2026-07-05T09:51:28.478495Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:51:28.478495Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Pareto Optimal Learning for Estimating Large Language Model Errors","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["stat.ML"],"primary_cat":"cs.CL","authors_text":"Hoifung Poon, J. Samuel Preston, Mu Wei, Theodore Zhao","submitted_at":"2023-06-28T21:11:15Z","abstract_excerpt":"Large Language Models (LLMs) have shown impressive abilities in many applications. When a concrete and precise answer is desired, it is important to have a quantitative estimation of the potential error rate. However, this can be challenging due to the text-in-text-out nature of generative models. We present a method based on Pareto optimization that generates a risk score to estimate the probability of error in an LLM response by integrating multiple sources of information. We prove theoretically that the error estimator optimized in our framework aligns with the LLM and the information sourc"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2306.16564","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2306.16564/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2306.16564","created_at":"2026-07-05T09:51:28.478554+00:00"},{"alias_kind":"arxiv_version","alias_value":"2306.16564v4","created_at":"2026-07-05T09:51:28.478554+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2306.16564","created_at":"2026-07-05T09:51:28.478554+00:00"},{"alias_kind":"pith_short_12","alias_value":"DWHCR4ZCW2RN","created_at":"2026-07-05T09:51:28.478554+00:00"},{"alias_kind":"pith_short_16","alias_value":"DWHCR4ZCW2RNEHGR","created_at":"2026-07-05T09:51:28.478554+00:00"},{"alias_kind":"pith_short_8","alias_value":"DWHCR4ZC","created_at":"2026-07-05T09:51:28.478554+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.04661","citing_title":"CRAFT: Cost-aware Refinement And Front-aware Tuning of Prompts","ref_index":173,"is_internal_anchor":false},{"citing_arxiv_id":"2604.17200","citing_title":"Calibrating Model-Based Evaluation Metrics for Summarization","ref_index":109,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/DWHCR4ZCW2RNEHGRSWAIULRYM3","json":"https://pith.science/pith/DWHCR4ZCW2RNEHGRSWAIULRYM3.json","graph_json":"https://pith.science/api/pith-number/DWHCR4ZCW2RNEHGRSWAIULRYM3/graph.json","events_json":"https://pith.science/api/pith-number/DWHCR4ZCW2RNEHGRSWAIULRYM3/events.json","paper":"https://pith.science/paper/DWHCR4ZC"},"agent_actions":{"view_html":"https://pith.science/pith/DWHCR4ZCW2RNEHGRSWAIULRYM3","download_json":"https://pith.science/pith/DWHCR4ZCW2RNEHGRSWAIULRYM3.json","view_paper":"https://pith.science/paper/DWHCR4ZC","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2306.16564&json=true","fetch_graph":"https://pith.science/api/pith-number/DWHCR4ZCW2RNEHGRSWAIULRYM3/graph.json","fetch_events":"https://pith.science/api/pith-number/DWHCR4ZCW2RNEHGRSWAIULRYM3/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/DWHCR4ZCW2RNEHGRSWAIULRYM3/action/timestamp_anchor","attest_storage":"https://pith.science/pith/DWHCR4ZCW2RNEHGRSWAIULRYM3/action/storage_attestation","attest_author":"https://pith.science/pith/DWHCR4ZCW2RNEHGRSWAIULRYM3/action/author_attestation","sign_citation":"https://pith.science/pith/DWHCR4ZCW2RNEHGRSWAIULRYM3/action/citation_signature","submit_replication":"https://pith.science/pith/DWHCR4ZCW2RNEHGRSWAIULRYM3/action/replication_record"}},"created_at":"2026-07-05T09:51:28.478554+00:00","updated_at":"2026-07-05T09:51:28.478554+00:00"}