{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:XAYI6NFWEY6RJ3O32EPQD6IPNJ","short_pith_number":"pith:XAYI6NFW","schema_version":"1.0","canonical_sha256":"b8308f34b6263d14eddbd11f01f90f6a45fafd7ee6455ebf7a4dafb6917f0fe5","source":{"kind":"arxiv","id":"2307.15343","version":2},"attestation_state":"computed","paper":{"title":"Med-HALT: Medical Domain Hallucination Test for Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG","stat.ML"],"primary_cat":"cs.CL","authors_text":"Ankit Pal, Logesh Kumar Umapathi, Malaikannan Sankarasubbu","submitted_at":"2023-07-28T06:43:04Z","abstract_excerpt":"This research paper focuses on the challenges posed by hallucinations in large language models (LLMs), particularly in the context of the medical domain. Hallucination, wherein these models generate plausible yet unverified or incorrect information, can have serious consequences in healthcare applications. We propose a new benchmark and dataset, Med-HALT (Medical Domain Hallucination Test), designed specifically to evaluate and reduce hallucinations. Med-HALT provides a diverse multinational dataset derived from medical examinations across various countries and includes multiple innovative tes"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2307.15343","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2023-07-28T06:43:04Z","cross_cats_sorted":["cs.AI","cs.LG","stat.ML"],"title_canon_sha256":"59dfbde1d7fbc5194b055c057dc612bd5f21a32de778e0ee3545853d2bc5888e","abstract_canon_sha256":"e2d079ff87c3955b6a95c9ab2ab8f7bde93b518c4596faefd4a6d1ac2f106c89"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:01:04.082778Z","signature_b64":"AZSb96NK/mhHx35+36wO+0Ll7TuPsOLpXPo4OsIGMreegrKzbtEwRXjYI0MSoY9andmuoFFP1stmttMdr+q3Aw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b8308f34b6263d14eddbd11f01f90f6a45fafd7ee6455ebf7a4dafb6917f0fe5","last_reissued_at":"2026-07-05T07:01:04.082399Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:01:04.082399Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Med-HALT: Medical Domain Hallucination Test for Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG","stat.ML"],"primary_cat":"cs.CL","authors_text":"Ankit Pal, Logesh Kumar Umapathi, Malaikannan Sankarasubbu","submitted_at":"2023-07-28T06:43:04Z","abstract_excerpt":"This research paper focuses on the challenges posed by hallucinations in large language models (LLMs), particularly in the context of the medical domain. Hallucination, wherein these models generate plausible yet unverified or incorrect information, can have serious consequences in healthcare applications. We propose a new benchmark and dataset, Med-HALT (Medical Domain Hallucination Test), designed specifically to evaluate and reduce hallucinations. Med-HALT provides a diverse multinational dataset derived from medical examinations across various countries and includes multiple innovative tes"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2307.15343","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2307.15343/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2307.15343","created_at":"2026-07-05T07:01:04.082455+00:00"},{"alias_kind":"arxiv_version","alias_value":"2307.15343v2","created_at":"2026-07-05T07:01:04.082455+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2307.15343","created_at":"2026-07-05T07:01:04.082455+00:00"},{"alias_kind":"pith_short_12","alias_value":"XAYI6NFWEY6R","created_at":"2026-07-05T07:01:04.082455+00:00"},{"alias_kind":"pith_short_16","alias_value":"XAYI6NFWEY6RJ3O3","created_at":"2026-07-05T07:01:04.082455+00:00"},{"alias_kind":"pith_short_8","alias_value":"XAYI6NFW","created_at":"2026-07-05T07:01:04.082455+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":12,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2606.07521","citing_title":"Evaluating Hallucinations in Domain-Adapted Large Language Models","ref_index":19,"is_internal_anchor":true},{"citing_arxiv_id":"2606.21517","citing_title":"MedHal-Loc: Are \"Explainable-by-Architecture\" Medical Hallucination Detectors Faithful Localizers? A Localization Benchmark","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2606.05436","citing_title":"Ten Headache Specialists versus Artificial Intelligence for Clinical Literature Summarization: A Critical Evaluation and Comparison","ref_index":73,"is_internal_anchor":false},{"citing_arxiv_id":"2606.01301","citing_title":"Med-HEAL: Analyzing and Mitigating Hallucinations in Medical LLMs with Hallucination-Aware In-Context Learning","ref_index":40,"is_internal_anchor":false},{"citing_arxiv_id":"2401.02458","citing_title":"Data-Centric Foundation Models in Computational Healthcare: A Survey","ref_index":293,"is_internal_anchor":false},{"citing_arxiv_id":"2412.14751","citing_title":"Query pipeline optimization for cancer patient question answering systems","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2605.20591","citing_title":"Do No Harm? Hallucination and Actor-Level Abuse in Web-Deployed Medical Large Language Models","ref_index":47,"is_internal_anchor":false},{"citing_arxiv_id":"2309.05922","citing_title":"A Survey of Hallucination in Large Foundation Models","ref_index":143,"is_internal_anchor":false},{"citing_arxiv_id":"2404.18416","citing_title":"Capabilities of Gemini Models in Medicine","ref_index":263,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08094","citing_title":"MedThink: Enhancing Diagnostic Accuracy in Small Models via Teacher-Guided Reasoning Correction","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2604.22843","citing_title":"Structure Guided Retrieval-Augmented Generation for Factual Queries","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2604.14829","citing_title":"Beyond Literal Summarization: Redefining Hallucination for Medical SOAP Note Evaluation","ref_index":8,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/XAYI6NFWEY6RJ3O32EPQD6IPNJ","json":"https://pith.science/pith/XAYI6NFWEY6RJ3O32EPQD6IPNJ.json","graph_json":"https://pith.science/api/pith-number/XAYI6NFWEY6RJ3O32EPQD6IPNJ/graph.json","events_json":"https://pith.science/api/pith-number/XAYI6NFWEY6RJ3O32EPQD6IPNJ/events.json","paper":"https://pith.science/paper/XAYI6NFW"},"agent_actions":{"view_html":"https://pith.science/pith/XAYI6NFWEY6RJ3O32EPQD6IPNJ","download_json":"https://pith.science/pith/XAYI6NFWEY6RJ3O32EPQD6IPNJ.json","view_paper":"https://pith.science/paper/XAYI6NFW","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2307.15343&json=true","fetch_graph":"https://pith.science/api/pith-number/XAYI6NFWEY6RJ3O32EPQD6IPNJ/graph.json","fetch_events":"https://pith.science/api/pith-number/XAYI6NFWEY6RJ3O32EPQD6IPNJ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/XAYI6NFWEY6RJ3O32EPQD6IPNJ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/XAYI6NFWEY6RJ3O32EPQD6IPNJ/action/storage_attestation","attest_author":"https://pith.science/pith/XAYI6NFWEY6RJ3O32EPQD6IPNJ/action/author_attestation","sign_citation":"https://pith.science/pith/XAYI6NFWEY6RJ3O32EPQD6IPNJ/action/citation_signature","submit_replication":"https://pith.science/pith/XAYI6NFWEY6RJ3O32EPQD6IPNJ/action/replication_record"}},"created_at":"2026-07-05T07:01:04.082455+00:00","updated_at":"2026-07-05T07:01:04.082455+00:00"}