{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:6RFKOCHZ2VWAAKK7BSVAJXCCYP","short_pith_number":"pith:6RFKOCHZ","schema_version":"1.0","canonical_sha256":"f44aa708f9d56c00295f0caa04dc42c3de2ce6f63402cbceeb7418ebcfcd7bf2","source":{"kind":"arxiv","id":"2503.06709","version":1},"attestation_state":"computed","paper":{"title":"Delusions of Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Hongshen Xu, Kai Yu, Kunyao Lan, Lu Chen, Mengyue Wu, Pascale Fung, Zichen Zhu, Zihan Wang, Ziwei Ji, Zixv yang","submitted_at":"2025-03-09T17:59:16Z","abstract_excerpt":"Large Language Models often generate factually incorrect but plausible outputs, known as hallucinations. We identify a more insidious phenomenon, LLM delusion, defined as high belief hallucinations, incorrect outputs with abnormally high confidence, making them harder to detect and mitigate. Unlike ordinary hallucinations, delusions persist with low uncertainty, posing significant challenges to model reliability. Through empirical analysis across different model families and sizes on several Question Answering tasks, we show that delusions are prevalent and distinct from hallucinations. LLMs e"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2503.06709","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2025-03-09T17:59:16Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"b2e4105fd3fb6212e5aa7ddb32896d085730c4bce9d4c4f2b4c2e0d21ba64e85","abstract_canon_sha256":"0b6443417bafbbb4d0d0498e092b3a31bde3ce58a314c6208a5b3938e6edbfc7"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:27:40.581518Z","signature_b64":"peRO9wIY22BOwIUZMD5EOL7zoLb5f2TGd6aq86ttXhOxh91pc/+lDo69gJQ26oJCowhAO4OkEQvXt7pfMsidBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f44aa708f9d56c00295f0caa04dc42c3de2ce6f63402cbceeb7418ebcfcd7bf2","last_reissued_at":"2026-07-05T10:27:40.580782Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:27:40.580782Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Delusions of Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Hongshen Xu, Kai Yu, Kunyao Lan, Lu Chen, Mengyue Wu, Pascale Fung, Zichen Zhu, Zihan Wang, Ziwei Ji, Zixv yang","submitted_at":"2025-03-09T17:59:16Z","abstract_excerpt":"Large Language Models often generate factually incorrect but plausible outputs, known as hallucinations. We identify a more insidious phenomenon, LLM delusion, defined as high belief hallucinations, incorrect outputs with abnormally high confidence, making them harder to detect and mitigate. Unlike ordinary hallucinations, delusions persist with low uncertainty, posing significant challenges to model reliability. Through empirical analysis across different model families and sizes on several Question Answering tasks, we show that delusions are prevalent and distinct from hallucinations. LLMs e"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2503.06709","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2503.06709/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2503.06709","created_at":"2026-07-05T10:27:40.580871+00:00"},{"alias_kind":"arxiv_version","alias_value":"2503.06709v1","created_at":"2026-07-05T10:27:40.580871+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2503.06709","created_at":"2026-07-05T10:27:40.580871+00:00"},{"alias_kind":"pith_short_12","alias_value":"6RFKOCHZ2VWA","created_at":"2026-07-05T10:27:40.580871+00:00"},{"alias_kind":"pith_short_16","alias_value":"6RFKOCHZ2VWAAKK7","created_at":"2026-07-05T10:27:40.580871+00:00"},{"alias_kind":"pith_short_8","alias_value":"6RFKOCHZ","created_at":"2026-07-05T10:27:40.580871+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.22007","citing_title":"Hallucination as Commitment Failure: Larger LLMs Misfire Despite Knowing the Answer","ref_index":15,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/6RFKOCHZ2VWAAKK7BSVAJXCCYP","json":"https://pith.science/pith/6RFKOCHZ2VWAAKK7BSVAJXCCYP.json","graph_json":"https://pith.science/api/pith-number/6RFKOCHZ2VWAAKK7BSVAJXCCYP/graph.json","events_json":"https://pith.science/api/pith-number/6RFKOCHZ2VWAAKK7BSVAJXCCYP/events.json","paper":"https://pith.science/paper/6RFKOCHZ"},"agent_actions":{"view_html":"https://pith.science/pith/6RFKOCHZ2VWAAKK7BSVAJXCCYP","download_json":"https://pith.science/pith/6RFKOCHZ2VWAAKK7BSVAJXCCYP.json","view_paper":"https://pith.science/paper/6RFKOCHZ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2503.06709&json=true","fetch_graph":"https://pith.science/api/pith-number/6RFKOCHZ2VWAAKK7BSVAJXCCYP/graph.json","fetch_events":"https://pith.science/api/pith-number/6RFKOCHZ2VWAAKK7BSVAJXCCYP/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/6RFKOCHZ2VWAAKK7BSVAJXCCYP/action/timestamp_anchor","attest_storage":"https://pith.science/pith/6RFKOCHZ2VWAAKK7BSVAJXCCYP/action/storage_attestation","attest_author":"https://pith.science/pith/6RFKOCHZ2VWAAKK7BSVAJXCCYP/action/author_attestation","sign_citation":"https://pith.science/pith/6RFKOCHZ2VWAAKK7BSVAJXCCYP/action/citation_signature","submit_replication":"https://pith.science/pith/6RFKOCHZ2VWAAKK7BSVAJXCCYP/action/replication_record"}},"created_at":"2026-07-05T10:27:40.580871+00:00","updated_at":"2026-07-05T10:27:40.580871+00:00"}