{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2025:LSSERFJPPJQQKLDQJTBOK2J65X","short_pith_number":"pith:LSSERFJP","canonical_record":{"source":{"id":"2509.04664","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-09-04T21:26:31Z","cross_cats_sorted":[],"title_canon_sha256":"3d51d6798cd9f5574f047632f5672d4a7b662df40ceeb4b4ee3daca282e445d3","abstract_canon_sha256":"c6442e3b6d4eb20e93b94d33f9e10fd927dcc669aff81ea05eae3accaf80e6a4"},"schema_version":"1.0"},"canonical_sha256":"5ca448952f7a61052c704cc2e5693eede6d03f33b526e0a04a01817442b622f4","source":{"kind":"arxiv","id":"2509.04664","version":1},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2509.04664","created_at":"2026-07-05T12:05:27Z"},{"alias_kind":"arxiv_version","alias_value":"2509.04664v1","created_at":"2026-07-05T12:05:27Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2509.04664","created_at":"2026-07-05T12:05:27Z"},{"alias_kind":"pith_short_12","alias_value":"LSSERFJPPJQQ","created_at":"2026-07-05T12:05:27Z"},{"alias_kind":"pith_short_16","alias_value":"LSSERFJPPJQQKLDQ","created_at":"2026-07-05T12:05:27Z"},{"alias_kind":"pith_short_8","alias_value":"LSSERFJP","created_at":"2026-07-05T12:05:27Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2025:LSSERFJPPJQQKLDQJTBOK2J65X","target":"record","payload":{"canonical_record":{"source":{"id":"2509.04664","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-09-04T21:26:31Z","cross_cats_sorted":[],"title_canon_sha256":"3d51d6798cd9f5574f047632f5672d4a7b662df40ceeb4b4ee3daca282e445d3","abstract_canon_sha256":"c6442e3b6d4eb20e93b94d33f9e10fd927dcc669aff81ea05eae3accaf80e6a4"},"schema_version":"1.0"},"canonical_sha256":"5ca448952f7a61052c704cc2e5693eede6d03f33b526e0a04a01817442b622f4","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T12:05:27.242355Z","signature_b64":"AuXBZZ37qHe+V5RzvfOfJ3i6Bglc+dNxbmk+J1+Y2xOt2jMKpnNDWXozl1MQScgkgI3BoqTuxzmYxM26RlltCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"5ca448952f7a61052c704cc2e5693eede6d03f33b526e0a04a01817442b622f4","last_reissued_at":"2026-07-05T12:05:27.241848Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T12:05:27.241848Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2509.04664","source_version":1,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T12:05:27Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"p5k3Gi1JX7XLpPc8yBzv0B3Xr2QxUgVb09dpy+POTu2zAraEIIcgLX50fWeBioj7H6uzyLd496ASHd7hgzW5Cg==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-06T01:48:02.031656Z"},"content_sha256":"b310982e1d28d3725676e216bb09806062e46a388f4b397d582e65266bbff7aa","schema_version":"1.0","event_id":"sha256:b310982e1d28d3725676e216bb09806062e46a388f4b397d582e65266bbff7aa"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2025:LSSERFJPPJQQKLDQJTBOK2J65X","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Why Language Models Hallucinate","license":"http://creativecommons.org/licenses/by/4.0/","headline":"Language models hallucinate because training and evaluations reward guessing when uncertain instead of admitting limits.","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Adam Tauman Kalai, Edwin Zhang, Ofir Nachum, Santosh S. Vempala","submitted_at":"2025-09-04T21:26:31Z","abstract_excerpt":"Like students facing hard exam questions, large language models sometimes guess when uncertain, producing plausible yet incorrect statements instead of admitting uncertainty. Such \"hallucinations\" persist even in state-of-the-art systems and undermine trust. We argue that language models hallucinate because the training and evaluation procedures reward guessing over acknowledging uncertainty, and we analyze the statistical causes of hallucinations in the modern training pipeline. Hallucinations need not be mysterious -- they originate simply as errors in binary classification. If incorrect sta"},"claims":{"count":4,"items":[{"kind":"strongest_claim","text":"We argue that language models hallucinate because the training and evaluation procedures reward guessing over acknowledging uncertainty, and we analyze the statistical causes of hallucinations in the modern training pipeline.","source":"verdict.strongest_claim","status":"machine_extracted","claim_id":"C1","attestation":"unclaimed"},{"kind":"weakest_assumption","text":"That hallucinations arise simply as errors in binary classification when incorrect statements cannot be distinguished from facts, and that modifying benchmark scoring will address the issue without introducing new problems.","source":"verdict.weakest_assumption","status":"machine_extracted","claim_id":"C2","attestation":"unclaimed"},{"kind":"one_line_summary","text":"Language models hallucinate due to statistical pressures from training pipelines that treat uncertainty as binary classification errors and from evaluations that penalize admitting ignorance.","source":"verdict.one_line_summary","status":"machine_extracted","claim_id":"C3","attestation":"unclaimed"},{"kind":"headline","text":"Language models hallucinate because training and evaluations reward guessing when uncertain instead of admitting limits.","source":"verdict.pith_extraction.headline","status":"machine_extracted","claim_id":"C4","attestation":"unclaimed"}],"snapshot_sha256":"3901295e1ee8c96e8363c4e1f3119a79af7f63a6eec5df7c1e3b200cec16a359"},"source":{"id":"2509.04664","kind":"arxiv","version":1},"verdict":{"id":"e89d876a-16ab-4fef-8bd1-024ac2bef5e7","model_set":{"reader":"grok-4.3"},"created_at":"2026-05-13T12:28:06.229364Z","strongest_claim":"We argue that language models hallucinate because the training and evaluation procedures reward guessing over acknowledging uncertainty, and we analyze the statistical causes of hallucinations in the modern training pipeline.","one_line_summary":"Language models hallucinate due to statistical pressures from training pipelines that treat uncertainty as binary classification errors and from evaluations that penalize admitting ignorance.","pipeline_version":"pith-pipeline@v0.9.0","weakest_assumption":"That hallucinations arise simply as errors in binary classification when incorrect statements cannot be distinguished from facts, and that modifying benchmark scoring will address the issue without introducing new problems.","pith_extraction_headline":"Language models hallucinate because training and evaluations reward guessing when uncertain instead of admitting limits."},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2509.04664/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":2,"sample":[{"doi":"10.18653/v1/2025.acl-long.1176","year":1962,"title":"Constitutional AI: Harmlessness from AI Feedback","work_id":"faaaa4e0-2676-4fac-a0b4-99aef10d2095","ref_index":1,"cited_arxiv_id":"2212.08073","is_internal_anchor":true},{"doi":"10.1145/3618260.3649777","year":2022,"title":"Language Models (Mostly) Know What They Know","work_id":"8ca58a10-da41-4f70-baae-7e449512e345","ref_index":2,"cited_arxiv_id":"2207.05221","is_internal_anchor":true}],"resolved_work":2,"snapshot_sha256":"81d2da1b5fc8067e4dcef8cd58a1a5fcb4cc0aeab9e5955186a629ca2c160cc0","internal_anchors":2},"formal_canon":{"evidence_count":3,"snapshot_sha256":"8891bcd4a8b698473872aaf36bb12b3594933012f19a4551e2c5aa9f2c3ed81a"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":"e89d876a-16ab-4fef-8bd1-024ac2bef5e7"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T12:05:27Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"5cbIS3jtWnjqusibwyA86YUbqVVKF2lylJYt1/116RGaZV/Df6LhZDTFxEenw1fYtOpRx/0Cbr5TqEOaiLsVDw==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-06T01:48:02.032375Z"},"content_sha256":"bc3df962062e2a115c1ed665c5b1fec6e04971b1354d9a9cc0247ccf45307f3c","schema_version":"1.0","event_id":"sha256:bc3df962062e2a115c1ed665c5b1fec6e04971b1354d9a9cc0247ccf45307f3c"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/LSSERFJPPJQQKLDQJTBOK2J65X/bundle.json","state_url":"https://pith.science/pith/LSSERFJPPJQQKLDQJTBOK2J65X/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/LSSERFJPPJQQKLDQJTBOK2J65X/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-06T01:48:02Z","links":{"resolver":"https://pith.science/pith/LSSERFJPPJQQKLDQJTBOK2J65X","bundle":"https://pith.science/pith/LSSERFJPPJQQKLDQJTBOK2J65X/bundle.json","state":"https://pith.science/pith/LSSERFJPPJQQKLDQJTBOK2J65X/state.json","well_known_bundle":"https://pith.science/.well-known/pith/LSSERFJPPJQQKLDQJTBOK2J65X/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2025:LSSERFJPPJQQKLDQJTBOK2J65X","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"c6442e3b6d4eb20e93b94d33f9e10fd927dcc669aff81ea05eae3accaf80e6a4","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-09-04T21:26:31Z","title_canon_sha256":"3d51d6798cd9f5574f047632f5672d4a7b662df40ceeb4b4ee3daca282e445d3"},"schema_version":"1.0","source":{"id":"2509.04664","kind":"arxiv","version":1}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2509.04664","created_at":"2026-07-05T12:05:27Z"},{"alias_kind":"arxiv_version","alias_value":"2509.04664v1","created_at":"2026-07-05T12:05:27Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2509.04664","created_at":"2026-07-05T12:05:27Z"},{"alias_kind":"pith_short_12","alias_value":"LSSERFJPPJQQ","created_at":"2026-07-05T12:05:27Z"},{"alias_kind":"pith_short_16","alias_value":"LSSERFJPPJQQKLDQ","created_at":"2026-07-05T12:05:27Z"},{"alias_kind":"pith_short_8","alias_value":"LSSERFJP","created_at":"2026-07-05T12:05:27Z"}],"graph_snapshots":[{"event_id":"sha256:bc3df962062e2a115c1ed665c5b1fec6e04971b1354d9a9cc0247ccf45307f3c","target":"graph","created_at":"2026-07-05T12:05:27Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":4,"items":[{"attestation":"unclaimed","claim_id":"C1","kind":"strongest_claim","source":"verdict.strongest_claim","status":"machine_extracted","text":"We argue that language models hallucinate because the training and evaluation procedures reward guessing over acknowledging uncertainty, and we analyze the statistical causes of hallucinations in the modern training pipeline."},{"attestation":"unclaimed","claim_id":"C2","kind":"weakest_assumption","source":"verdict.weakest_assumption","status":"machine_extracted","text":"That hallucinations arise simply as errors in binary classification when incorrect statements cannot be distinguished from facts, and that modifying benchmark scoring will address the issue without introducing new problems."},{"attestation":"unclaimed","claim_id":"C3","kind":"one_line_summary","source":"verdict.one_line_summary","status":"machine_extracted","text":"Language models hallucinate due to statistical pressures from training pipelines that treat uncertainty as binary classification errors and from evaluations that penalize admitting ignorance."},{"attestation":"unclaimed","claim_id":"C4","kind":"headline","source":"verdict.pith_extraction.headline","status":"machine_extracted","text":"Language models hallucinate because training and evaluations reward guessing when uncertain instead of admitting limits."}],"snapshot_sha256":"3901295e1ee8c96e8363c4e1f3119a79af7f63a6eec5df7c1e3b200cec16a359"},"formal_canon":{"evidence_count":3,"snapshot_sha256":"8891bcd4a8b698473872aaf36bb12b3594933012f19a4551e2c5aa9f2c3ed81a"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2509.04664/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Like students facing hard exam questions, large language models sometimes guess when uncertain, producing plausible yet incorrect statements instead of admitting uncertainty. Such \"hallucinations\" persist even in state-of-the-art systems and undermine trust. We argue that language models hallucinate because the training and evaluation procedures reward guessing over acknowledging uncertainty, and we analyze the statistical causes of hallucinations in the modern training pipeline. Hallucinations need not be mysterious -- they originate simply as errors in binary classification. If incorrect sta","authors_text":"Adam Tauman Kalai, Edwin Zhang, Ofir Nachum, Santosh S. Vempala","cross_cats":[],"headline":"Language models hallucinate because training and evaluations reward guessing when uncertain instead of admitting limits.","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-09-04T21:26:31Z","title":"Why Language Models Hallucinate"},"references":{"count":2,"internal_anchors":2,"resolved_work":2,"sample":[{"cited_arxiv_id":"2212.08073","doi":"10.18653/v1/2025.acl-long.1176","is_internal_anchor":true,"ref_index":1,"title":"Constitutional AI: Harmlessness from AI Feedback","work_id":"faaaa4e0-2676-4fac-a0b4-99aef10d2095","year":1962},{"cited_arxiv_id":"2207.05221","doi":"10.1145/3618260.3649777","is_internal_anchor":true,"ref_index":2,"title":"Language Models (Mostly) Know What They Know","work_id":"8ca58a10-da41-4f70-baae-7e449512e345","year":2022}],"snapshot_sha256":"81d2da1b5fc8067e4dcef8cd58a1a5fcb4cc0aeab9e5955186a629ca2c160cc0"},"source":{"id":"2509.04664","kind":"arxiv","version":1},"verdict":{"created_at":"2026-05-13T12:28:06.229364Z","id":"e89d876a-16ab-4fef-8bd1-024ac2bef5e7","model_set":{"reader":"grok-4.3"},"one_line_summary":"Language models hallucinate due to statistical pressures from training pipelines that treat uncertainty as binary classification errors and from evaluations that penalize admitting ignorance.","pipeline_version":"pith-pipeline@v0.9.0","pith_extraction_headline":"Language models hallucinate because training and evaluations reward guessing when uncertain instead of admitting limits.","strongest_claim":"We argue that language models hallucinate because the training and evaluation procedures reward guessing over acknowledging uncertainty, and we analyze the statistical causes of hallucinations in the modern training pipeline.","weakest_assumption":"That hallucinations arise simply as errors in binary classification when incorrect statements cannot be distinguished from facts, and that modifying benchmark scoring will address the issue without introducing new problems."}},"verdict_id":"e89d876a-16ab-4fef-8bd1-024ac2bef5e7"}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:b310982e1d28d3725676e216bb09806062e46a388f4b397d582e65266bbff7aa","target":"record","created_at":"2026-07-05T12:05:27Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"c6442e3b6d4eb20e93b94d33f9e10fd927dcc669aff81ea05eae3accaf80e6a4","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-09-04T21:26:31Z","title_canon_sha256":"3d51d6798cd9f5574f047632f5672d4a7b662df40ceeb4b4ee3daca282e445d3"},"schema_version":"1.0","source":{"id":"2509.04664","kind":"arxiv","version":1}},"canonical_sha256":"5ca448952f7a61052c704cc2e5693eede6d03f33b526e0a04a01817442b622f4","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"5ca448952f7a61052c704cc2e5693eede6d03f33b526e0a04a01817442b622f4","first_computed_at":"2026-07-05T12:05:27.241848Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T12:05:27.241848Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"AuXBZZ37qHe+V5RzvfOfJ3i6Bglc+dNxbmk+J1+Y2xOt2jMKpnNDWXozl1MQScgkgI3BoqTuxzmYxM26RlltCg==","signature_status":"signed_v1","signed_at":"2026-07-05T12:05:27.242355Z","signed_message":"canonical_sha256_bytes"},"source_id":"2509.04664","source_kind":"arxiv","source_version":1}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:b310982e1d28d3725676e216bb09806062e46a388f4b397d582e65266bbff7aa","sha256:bc3df962062e2a115c1ed665c5b1fec6e04971b1354d9a9cc0247ccf45307f3c"],"state_sha256":"cd0dc690b60dc9d96f101e2d388e021b60d3532760b2d2df2f16440e82969104"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"tKq7TalCWlQTpherjAcKYRHegHo1MfLkrpcLTFcwNpNkMD4DU8h3z1vFHHLST4s2tQcBU+OQnLOwZtwaJH+fCA==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-06T01:48:02.037861Z","bundle_sha256":"a9b5e8975bd5d87d793aa77d3b59607e2c3cbf6375701e32bf8896a3a19f1970"}}