{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:GD226NULWLTG4OUKEYIO3M4OGM","short_pith_number":"pith:GD226NUL","schema_version":"1.0","canonical_sha256":"30f5af368bb2e66e3a8a2610edb38e3337f935ae937b8d22993f45610ae8eb7a","source":{"kind":"arxiv","id":"2404.00474","version":2},"attestation_state":"computed","paper":{"title":"Linguistic Calibration of Long-Form Generations","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL","stat.ML"],"primary_cat":"cs.LG","authors_text":"Neil Band, Tatsunori Hashimoto, Tengyu Ma, Xuechen Li","submitted_at":"2024-03-30T20:47:55Z","abstract_excerpt":"Language models (LMs) may lead their users to make suboptimal downstream decisions when they confidently hallucinate. This issue can be mitigated by having the LM verbally convey the probability that its claims are correct, but existing models cannot produce long-form text with calibrated confidence statements. Through the lens of decision-making, we define linguistic calibration for long-form generations: an LM is linguistically calibrated if its generations enable its users to make calibrated probabilistic predictions. This definition enables a training framework where a supervised finetunin"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2404.00474","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-03-30T20:47:55Z","cross_cats_sorted":["cs.AI","cs.CL","stat.ML"],"title_canon_sha256":"ac8e4a34eca49d0d1aa592730d55d16483a8aacc52b3d05f258c763fcd03f27d","abstract_canon_sha256":"d48b842560e07f026967a9dfcdd40956bca80ab43d9c1def4d778a90016ff794"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:27:25.658541Z","signature_b64":"T/cdjkRLlURwavzNbBehHhxvtygm7F0kZ4DbKe1I4FhwHJHx5ebx4vIkWQS5/Ae9QRHa3RFT8HjbDFgZhif1Cw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"30f5af368bb2e66e3a8a2610edb38e3337f935ae937b8d22993f45610ae8eb7a","last_reissued_at":"2026-07-05T08:27:25.658040Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:27:25.658040Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Linguistic Calibration of Long-Form Generations","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL","stat.ML"],"primary_cat":"cs.LG","authors_text":"Neil Band, Tatsunori Hashimoto, Tengyu Ma, Xuechen Li","submitted_at":"2024-03-30T20:47:55Z","abstract_excerpt":"Language models (LMs) may lead their users to make suboptimal downstream decisions when they confidently hallucinate. This issue can be mitigated by having the LM verbally convey the probability that its claims are correct, but existing models cannot produce long-form text with calibrated confidence statements. Through the lens of decision-making, we define linguistic calibration for long-form generations: an LM is linguistically calibrated if its generations enable its users to make calibrated probabilistic predictions. This definition enables a training framework where a supervised finetunin"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2404.00474","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2404.00474/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2404.00474","created_at":"2026-07-05T08:27:25.658105+00:00"},{"alias_kind":"arxiv_version","alias_value":"2404.00474v2","created_at":"2026-07-05T08:27:25.658105+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2404.00474","created_at":"2026-07-05T08:27:25.658105+00:00"},{"alias_kind":"pith_short_12","alias_value":"GD226NULWLTG","created_at":"2026-07-05T08:27:25.658105+00:00"},{"alias_kind":"pith_short_16","alias_value":"GD226NULWLTG4OUK","created_at":"2026-07-05T08:27:25.658105+00:00"},{"alias_kind":"pith_short_8","alias_value":"GD226NUL","created_at":"2026-07-05T08:27:25.658105+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2607.00164","citing_title":"Verifiable Rewards for Calibrated Probabilistic Forecasting","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2606.32032","citing_title":"Reinforcement Learning with Metacognitive Feedback Elicits Faithful Uncertainty Expression in LLMs","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2605.29605","citing_title":"VLAConf: Calibrated Task-Success Confidence for Vision-Language-Action Models","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2605.28778","citing_title":"Can LLMs Use Linguistic Uncertainty Markers to Reliably Reflect Intrinsic Confidence?","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2406.15927","citing_title":"Semantic Entropy Probes: Robust and Cheap Hallucination Detection in LLMs","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10202","citing_title":"Task-Aware Calibration: Provably Optimal Decoding in LLMs","ref_index":1,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/GD226NULWLTG4OUKEYIO3M4OGM","json":"https://pith.science/pith/GD226NULWLTG4OUKEYIO3M4OGM.json","graph_json":"https://pith.science/api/pith-number/GD226NULWLTG4OUKEYIO3M4OGM/graph.json","events_json":"https://pith.science/api/pith-number/GD226NULWLTG4OUKEYIO3M4OGM/events.json","paper":"https://pith.science/paper/GD226NUL"},"agent_actions":{"view_html":"https://pith.science/pith/GD226NULWLTG4OUKEYIO3M4OGM","download_json":"https://pith.science/pith/GD226NULWLTG4OUKEYIO3M4OGM.json","view_paper":"https://pith.science/paper/GD226NUL","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2404.00474&json=true","fetch_graph":"https://pith.science/api/pith-number/GD226NULWLTG4OUKEYIO3M4OGM/graph.json","fetch_events":"https://pith.science/api/pith-number/GD226NULWLTG4OUKEYIO3M4OGM/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/GD226NULWLTG4OUKEYIO3M4OGM/action/timestamp_anchor","attest_storage":"https://pith.science/pith/GD226NULWLTG4OUKEYIO3M4OGM/action/storage_attestation","attest_author":"https://pith.science/pith/GD226NULWLTG4OUKEYIO3M4OGM/action/author_attestation","sign_citation":"https://pith.science/pith/GD226NULWLTG4OUKEYIO3M4OGM/action/citation_signature","submit_replication":"https://pith.science/pith/GD226NULWLTG4OUKEYIO3M4OGM/action/replication_record"}},"created_at":"2026-07-05T08:27:25.658105+00:00","updated_at":"2026-07-05T08:27:25.658105+00:00"}