{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:ZWILG2Q6MAHMK4FRXOOD7IPFJL","short_pith_number":"pith:ZWILG2Q6","schema_version":"1.0","canonical_sha256":"cd90b36a1e600ec570b1bb9c3fa1e54ad08ebaf5efc5ad971d494db21522503c","source":{"kind":"arxiv","id":"2402.10978","version":1},"attestation_state":"computed","paper":{"title":"Language Models with Conformal Factuality Guarantees","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Christopher Mohri, Tatsunori Hashimoto","submitted_at":"2024-02-15T18:31:53Z","abstract_excerpt":"Guaranteeing the correctness and factuality of language model (LM) outputs is a major open problem. In this work, we propose conformal factuality, a framework that can ensure high probability correctness guarantees for LMs by connecting language modeling and conformal prediction. We observe that the correctness of an LM output is equivalent to an uncertainty quantification problem, where the uncertainty sets are defined as the entailment set of an LM's output. Using this connection, we show that conformal prediction in language models corresponds to a back-off algorithm that provides high prob"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2402.10978","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-02-15T18:31:53Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"99122aa60baef36ebcb55216be1dd991a425c41372b4135c9a6282456b820e98","abstract_canon_sha256":"1f9250eedb288e36e055609624fa4abefb7d9c760c637fdb867f16298add2bde"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:46:03.155640Z","signature_b64":"tHdTr6OCP7pT5/K6eBM6dyLWkZwmJWJpx6HjfQ2UJ9wuvkY6x+VIXlQoSEK32i0LoTZ4ayMquaHK3EWZM+aQAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"cd90b36a1e600ec570b1bb9c3fa1e54ad08ebaf5efc5ad971d494db21522503c","last_reissued_at":"2026-07-05T07:46:03.155198Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:46:03.155198Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Language Models with Conformal Factuality Guarantees","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Christopher Mohri, Tatsunori Hashimoto","submitted_at":"2024-02-15T18:31:53Z","abstract_excerpt":"Guaranteeing the correctness and factuality of language model (LM) outputs is a major open problem. In this work, we propose conformal factuality, a framework that can ensure high probability correctness guarantees for LMs by connecting language modeling and conformal prediction. We observe that the correctness of an LM output is equivalent to an uncertainty quantification problem, where the uncertainty sets are defined as the entailment set of an LM's output. Using this connection, we show that conformal prediction in language models corresponds to a back-off algorithm that provides high prob"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2402.10978","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2402.10978/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2402.10978","created_at":"2026-07-05T07:46:03.155254+00:00"},{"alias_kind":"arxiv_version","alias_value":"2402.10978v1","created_at":"2026-07-05T07:46:03.155254+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2402.10978","created_at":"2026-07-05T07:46:03.155254+00:00"},{"alias_kind":"pith_short_12","alias_value":"ZWILG2Q6MAHM","created_at":"2026-07-05T07:46:03.155254+00:00"},{"alias_kind":"pith_short_16","alias_value":"ZWILG2Q6MAHMK4FR","created_at":"2026-07-05T07:46:03.155254+00:00"},{"alias_kind":"pith_short_8","alias_value":"ZWILG2Q6","created_at":"2026-07-05T07:46:03.155254+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":12,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2607.02510","citing_title":"Online Safety Monitoring for LLMs","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2606.12587","citing_title":"Strategic Decision Support for AI Agents","ref_index":52,"is_internal_anchor":false},{"citing_arxiv_id":"2607.02206","citing_title":"Prediction Sets for Counterfactual Decisions: Coverage, Optimality, and Conformal Prediction","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2606.07570","citing_title":"Can LLMs extract scientific consensus? A case study in high-temperature superconductivity","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2604.17487","citing_title":"Answer Only as Precisely as Justified: Calibrated Claim-Level Specificity Control for Agentic Systems","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18812","citing_title":"PASC: Pipeline-Aware Conformal Prediction with Joint Coverage Guarantees for Multi-Stage NLP and LLM Pipelines","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2605.15416","citing_title":"Margin-Adaptive Confidence Ranking for Reliable LLM Judgement","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12813","citing_title":"REALISTA: Realistic Latent Adversarial Attacks that Elicit LLM Hallucinations","ref_index":175,"is_internal_anchor":false},{"citing_arxiv_id":"2604.01413","citing_title":"Adaptive Stopping for Multi-Turn LLM Reasoning","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2604.27914","citing_title":"Geometry-Calibrated Conformal Abstention for Language Models","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2604.15109","citing_title":"IUQ: Interrogative Uncertainty Quantification for Long-Form Large Language Model Generation","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2604.17487","citing_title":"Answer Only as Precisely as Justified: Calibrated Claim-Level Specificity Control for Agentic Systems","ref_index":11,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ZWILG2Q6MAHMK4FRXOOD7IPFJL","json":"https://pith.science/pith/ZWILG2Q6MAHMK4FRXOOD7IPFJL.json","graph_json":"https://pith.science/api/pith-number/ZWILG2Q6MAHMK4FRXOOD7IPFJL/graph.json","events_json":"https://pith.science/api/pith-number/ZWILG2Q6MAHMK4FRXOOD7IPFJL/events.json","paper":"https://pith.science/paper/ZWILG2Q6"},"agent_actions":{"view_html":"https://pith.science/pith/ZWILG2Q6MAHMK4FRXOOD7IPFJL","download_json":"https://pith.science/pith/ZWILG2Q6MAHMK4FRXOOD7IPFJL.json","view_paper":"https://pith.science/paper/ZWILG2Q6","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2402.10978&json=true","fetch_graph":"https://pith.science/api/pith-number/ZWILG2Q6MAHMK4FRXOOD7IPFJL/graph.json","fetch_events":"https://pith.science/api/pith-number/ZWILG2Q6MAHMK4FRXOOD7IPFJL/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ZWILG2Q6MAHMK4FRXOOD7IPFJL/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ZWILG2Q6MAHMK4FRXOOD7IPFJL/action/storage_attestation","attest_author":"https://pith.science/pith/ZWILG2Q6MAHMK4FRXOOD7IPFJL/action/author_attestation","sign_citation":"https://pith.science/pith/ZWILG2Q6MAHMK4FRXOOD7IPFJL/action/citation_signature","submit_replication":"https://pith.science/pith/ZWILG2Q6MAHMK4FRXOOD7IPFJL/action/replication_record"}},"created_at":"2026-07-05T07:46:03.155254+00:00","updated_at":"2026-07-05T07:46:03.155254+00:00"}