{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:KLMBXVYKV4QV6H5KV5NX5PTZAC","short_pith_number":"pith:KLMBXVYK","schema_version":"1.0","canonical_sha256":"52d81bd70aaf215f1faaaf5b7ebe7900b66d900164bb785d79db7be8bdfc9414","source":{"kind":"arxiv","id":"2305.14552","version":2},"attestation_state":"computed","paper":{"title":"Sources of Hallucination by Large Language Models on Inference Tasks","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Liang Cheng, Mark Johnson, Mark Steedman, Mohammad Javad Hosseini, Nick McKenna, Tianyi Li","submitted_at":"2023-05-23T22:24:44Z","abstract_excerpt":"Large Language Models (LLMs) are claimed to be capable of Natural Language Inference (NLI), necessary for applied tasks like question answering and summarization. We present a series of behavioral studies on several LLM families (LLaMA, GPT-3.5, and PaLM) which probe their behavior using controlled experiments. We establish two biases originating from pretraining which predict much of their behavior, and show that these are major sources of hallucination in generative LLMs. First, memorization at the level of sentences: we show that, regardless of the premise, models falsely label NLI test sam"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2305.14552","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2023-05-23T22:24:44Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"2a21c4d5476faf9310aa847660c9e4b383599ea205be040e06bfd5133b0d4977","abstract_canon_sha256":"eb9e0d4494b3c3a326ab3b336e424b225789421eb9e8dbbc109c75f5d9288d9b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:03:25.524182Z","signature_b64":"RNU7Rny7wrHyOFZR3n/Ca4mt//I4TILzE8zwbSnxg9VxI4BrNZsm4VgHwdjKoiYEyVhrOvp87FcOGWSMKzV2AA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"52d81bd70aaf215f1faaaf5b7ebe7900b66d900164bb785d79db7be8bdfc9414","last_reissued_at":"2026-07-05T07:03:25.523698Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:03:25.523698Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Sources of Hallucination by Large Language Models on Inference Tasks","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Liang Cheng, Mark Johnson, Mark Steedman, Mohammad Javad Hosseini, Nick McKenna, Tianyi Li","submitted_at":"2023-05-23T22:24:44Z","abstract_excerpt":"Large Language Models (LLMs) are claimed to be capable of Natural Language Inference (NLI), necessary for applied tasks like question answering and summarization. We present a series of behavioral studies on several LLM families (LLaMA, GPT-3.5, and PaLM) which probe their behavior using controlled experiments. We establish two biases originating from pretraining which predict much of their behavior, and show that these are major sources of hallucination in generative LLMs. First, memorization at the level of sentences: we show that, regardless of the premise, models falsely label NLI test sam"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2305.14552","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2305.14552/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2305.14552","created_at":"2026-07-05T07:03:25.523751+00:00"},{"alias_kind":"arxiv_version","alias_value":"2305.14552v2","created_at":"2026-07-05T07:03:25.523751+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2305.14552","created_at":"2026-07-05T07:03:25.523751+00:00"},{"alias_kind":"pith_short_12","alias_value":"KLMBXVYKV4QV","created_at":"2026-07-05T07:03:25.523751+00:00"},{"alias_kind":"pith_short_16","alias_value":"KLMBXVYKV4QV6H5K","created_at":"2026-07-05T07:03:25.523751+00:00"},{"alias_kind":"pith_short_8","alias_value":"KLMBXVYK","created_at":"2026-07-05T07:03:25.523751+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2607.01612","citing_title":"Scaling with Confidence: Calibrating Confidence of LLMs for Adaptive Test Time Scaling","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2606.18285","citing_title":"RELIANCE: Curating and Evaluating Reproductive Health Information on Social Media","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2606.00819","citing_title":"Mitigating Hallucinations in Large Language Models Via Decoder Layer Skipping","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2503.18562","citing_title":"Self-Reported Confidence of Large Language Models in Gastroenterology: Analysis of Commercial, Open-Source, and Quantized Models","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2309.05922","citing_title":"A Survey of Hallucination in Large Foundation Models","ref_index":137,"is_internal_anchor":false},{"citing_arxiv_id":"2604.05467","citing_title":"CUE-R: Beyond the Final Answer in Retrieval-Augmented Generation","ref_index":20,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/KLMBXVYKV4QV6H5KV5NX5PTZAC","json":"https://pith.science/pith/KLMBXVYKV4QV6H5KV5NX5PTZAC.json","graph_json":"https://pith.science/api/pith-number/KLMBXVYKV4QV6H5KV5NX5PTZAC/graph.json","events_json":"https://pith.science/api/pith-number/KLMBXVYKV4QV6H5KV5NX5PTZAC/events.json","paper":"https://pith.science/paper/KLMBXVYK"},"agent_actions":{"view_html":"https://pith.science/pith/KLMBXVYKV4QV6H5KV5NX5PTZAC","download_json":"https://pith.science/pith/KLMBXVYKV4QV6H5KV5NX5PTZAC.json","view_paper":"https://pith.science/paper/KLMBXVYK","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2305.14552&json=true","fetch_graph":"https://pith.science/api/pith-number/KLMBXVYKV4QV6H5KV5NX5PTZAC/graph.json","fetch_events":"https://pith.science/api/pith-number/KLMBXVYKV4QV6H5KV5NX5PTZAC/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/KLMBXVYKV4QV6H5KV5NX5PTZAC/action/timestamp_anchor","attest_storage":"https://pith.science/pith/KLMBXVYKV4QV6H5KV5NX5PTZAC/action/storage_attestation","attest_author":"https://pith.science/pith/KLMBXVYKV4QV6H5KV5NX5PTZAC/action/author_attestation","sign_citation":"https://pith.science/pith/KLMBXVYKV4QV6H5KV5NX5PTZAC/action/citation_signature","submit_replication":"https://pith.science/pith/KLMBXVYKV4QV6H5KV5NX5PTZAC/action/replication_record"}},"created_at":"2026-07-05T07:03:25.523751+00:00","updated_at":"2026-07-05T07:03:25.523751+00:00"}