{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:OPJZ2DWLZ53T34PBEGDJJVQ4QX","short_pith_number":"pith:OPJZ2DWL","schema_version":"1.0","canonical_sha256":"73d39d0ecbcf773df1e1218694d61c85fda5172d277830da06eb2ad7e62b7f30","source":{"kind":"arxiv","id":"2305.18248","version":3},"attestation_state":"computed","paper":{"title":"Do Language Models Know When They're Hallucinating References?","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Adam Tauman Kalai, Ayush Agrawal, Lester Mackey, Mirac Suzgun","submitted_at":"2023-05-29T17:12:03Z","abstract_excerpt":"State-of-the-art language models (LMs) are notoriously susceptible to generating hallucinated information. Such inaccurate outputs not only undermine the reliability of these models but also limit their use and raise serious concerns about misinformation and propaganda. In this work, we focus on hallucinated book and article references and present them as the \"model organism\" of language model hallucination research, due to their frequent and easy-to-discern nature. We posit that if a language model cites a particular reference in its output, then it should ideally possess sufficient informati"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2305.18248","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by-sa/4.0/","primary_cat":"cs.CL","submitted_at":"2023-05-29T17:12:03Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"794873cb8195b659eff557bdded889a66b17a2e6ab0fa4ab656319e264fb4438","abstract_canon_sha256":"38ad6ca959d575ef03dab4b2155b78950386e52a6a7d91e70218aff5ce064057"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:58:20.422677Z","signature_b64":"/cK2P0n/2v/T1jQU3p8XIsoF/kg/hBpHg1FypIqNmldFHmf4RZKdkJa8PMvlmN/GQ4arrPfzklyzcsVCO6ehAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"73d39d0ecbcf773df1e1218694d61c85fda5172d277830da06eb2ad7e62b7f30","last_reissued_at":"2026-07-05T07:58:20.422186Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:58:20.422186Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Do Language Models Know When They're Hallucinating References?","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Adam Tauman Kalai, Ayush Agrawal, Lester Mackey, Mirac Suzgun","submitted_at":"2023-05-29T17:12:03Z","abstract_excerpt":"State-of-the-art language models (LMs) are notoriously susceptible to generating hallucinated information. Such inaccurate outputs not only undermine the reliability of these models but also limit their use and raise serious concerns about misinformation and propaganda. In this work, we focus on hallucinated book and article references and present them as the \"model organism\" of language model hallucination research, due to their frequent and easy-to-discern nature. We posit that if a language model cites a particular reference in its output, then it should ideally possess sufficient informati"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2305.18248","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2305.18248/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2305.18248","created_at":"2026-07-05T07:58:20.422247+00:00"},{"alias_kind":"arxiv_version","alias_value":"2305.18248v3","created_at":"2026-07-05T07:58:20.422247+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2305.18248","created_at":"2026-07-05T07:58:20.422247+00:00"},{"alias_kind":"pith_short_12","alias_value":"OPJZ2DWLZ53T","created_at":"2026-07-05T07:58:20.422247+00:00"},{"alias_kind":"pith_short_16","alias_value":"OPJZ2DWLZ53T34PB","created_at":"2026-07-05T07:58:20.422247+00:00"},{"alias_kind":"pith_short_8","alias_value":"OPJZ2DWL","created_at":"2026-07-05T07:58:20.422247+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2411.15594","citing_title":"A Survey on LLM-as-a-Judge","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2309.11495","citing_title":"Chain-of-Verification Reduces Hallucination in Large Language Models","ref_index":107,"is_internal_anchor":false},{"citing_arxiv_id":"2411.04368","citing_title":"Measuring short-form factuality in large language models","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2604.03159","citing_title":"BibTeX Citation Hallucinations in Scientific Publishing Agents: Evaluation and Mitigation","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2311.05232","citing_title":"A Survey on Hallucination in Large Language Models: Principles, Taxonomy, Challenges, and Open Questions","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2604.15945","citing_title":"RAGognizer: Hallucination-Aware Fine-Tuning via Detection Head Integration","ref_index":6,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/OPJZ2DWLZ53T34PBEGDJJVQ4QX","json":"https://pith.science/pith/OPJZ2DWLZ53T34PBEGDJJVQ4QX.json","graph_json":"https://pith.science/api/pith-number/OPJZ2DWLZ53T34PBEGDJJVQ4QX/graph.json","events_json":"https://pith.science/api/pith-number/OPJZ2DWLZ53T34PBEGDJJVQ4QX/events.json","paper":"https://pith.science/paper/OPJZ2DWL"},"agent_actions":{"view_html":"https://pith.science/pith/OPJZ2DWLZ53T34PBEGDJJVQ4QX","download_json":"https://pith.science/pith/OPJZ2DWLZ53T34PBEGDJJVQ4QX.json","view_paper":"https://pith.science/paper/OPJZ2DWL","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2305.18248&json=true","fetch_graph":"https://pith.science/api/pith-number/OPJZ2DWLZ53T34PBEGDJJVQ4QX/graph.json","fetch_events":"https://pith.science/api/pith-number/OPJZ2DWLZ53T34PBEGDJJVQ4QX/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/OPJZ2DWLZ53T34PBEGDJJVQ4QX/action/timestamp_anchor","attest_storage":"https://pith.science/pith/OPJZ2DWLZ53T34PBEGDJJVQ4QX/action/storage_attestation","attest_author":"https://pith.science/pith/OPJZ2DWLZ53T34PBEGDJJVQ4QX/action/author_attestation","sign_citation":"https://pith.science/pith/OPJZ2DWLZ53T34PBEGDJJVQ4QX/action/citation_signature","submit_replication":"https://pith.science/pith/OPJZ2DWLZ53T34PBEGDJJVQ4QX/action/replication_record"}},"created_at":"2026-07-05T07:58:20.422247+00:00","updated_at":"2026-07-05T07:58:20.422247+00:00"}