{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:LEKBR5LZ3IP3KI7Z6SYAIUSUFX","short_pith_number":"pith:LEKBR5LZ","schema_version":"1.0","canonical_sha256":"591418f579da1fb523f9f4b00452542df9f498728d0dac0ef405a7ae82d8dea1","source":{"kind":"arxiv","id":"2403.04696","version":2},"attestation_state":"computed","paper":{"title":"Fact-Checking the Output of Large Language Models via Token-Level Uncertainty Quantification","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Aleksandr Rubashevskii, Alexander Panchenko, Artem Shelmanov, Ekaterina Fadeeva, Evgenii Tsymbalov, Gleb Kuzmin, Hamdy Mubarak, Haonan Li, Maxim Panov, Preslav Nakov, Sergey Petrakov, Timothy Baldwin","submitted_at":"2024-03-07T17:44:17Z","abstract_excerpt":"Large language models (LLMs) are notorious for hallucinating, i.e., producing erroneous claims in their output. Such hallucinations can be dangerous, as occasional factual inaccuracies in the generated text might be obscured by the rest of the output being generally factually correct, making it extremely hard for the users to spot them. Current services that leverage LLMs usually do not provide any means for detecting unreliable generations. Here, we aim to bridge this gap. In particular, we propose a novel fact-checking and hallucination detection pipeline based on token-level uncertainty qua"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2403.04696","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-03-07T17:44:17Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"e56a79936056095339773806d4d4b57c084ebe125960064a7b3ce6dc00f3626c","abstract_canon_sha256":"fccc5687d8d2443a9f160e87e594fc787d1c99978e95e7e02bb219ca89a46c2f"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:28:39.364580Z","signature_b64":"rnKEBf5qszxoa59GYzxNmtUwOwtY7G1UTvtdUvQHY61uDHJC53NXDMDHp4wyXV56qB7PnaWuoGWtxMGL6DpiAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"591418f579da1fb523f9f4b00452542df9f498728d0dac0ef405a7ae82d8dea1","last_reissued_at":"2026-07-05T08:28:39.364134Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:28:39.364134Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Fact-Checking the Output of Large Language Models via Token-Level Uncertainty Quantification","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Aleksandr Rubashevskii, Alexander Panchenko, Artem Shelmanov, Ekaterina Fadeeva, Evgenii Tsymbalov, Gleb Kuzmin, Hamdy Mubarak, Haonan Li, Maxim Panov, Preslav Nakov, Sergey Petrakov, Timothy Baldwin","submitted_at":"2024-03-07T17:44:17Z","abstract_excerpt":"Large language models (LLMs) are notorious for hallucinating, i.e., producing erroneous claims in their output. Such hallucinations can be dangerous, as occasional factual inaccuracies in the generated text might be obscured by the rest of the output being generally factually correct, making it extremely hard for the users to spot them. Current services that leverage LLMs usually do not provide any means for detecting unreliable generations. Here, we aim to bridge this gap. In particular, we propose a novel fact-checking and hallucination detection pipeline based on token-level uncertainty qua"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2403.04696","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2403.04696/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2403.04696","created_at":"2026-07-05T08:28:39.364190+00:00"},{"alias_kind":"arxiv_version","alias_value":"2403.04696v2","created_at":"2026-07-05T08:28:39.364190+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2403.04696","created_at":"2026-07-05T08:28:39.364190+00:00"},{"alias_kind":"pith_short_12","alias_value":"LEKBR5LZ3IP3","created_at":"2026-07-05T08:28:39.364190+00:00"},{"alias_kind":"pith_short_16","alias_value":"LEKBR5LZ3IP3KI7Z","created_at":"2026-07-05T08:28:39.364190+00:00"},{"alias_kind":"pith_short_8","alias_value":"LEKBR5LZ","created_at":"2026-07-05T08:28:39.364190+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":11,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.29799","citing_title":"The CRISTAL Method: Neurosymbolic analysis from AI-synthesized world models","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2408.10692","citing_title":"Unconditional Truthfulness: Learning Unconditional Uncertainty of Large Language Models","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2502.14427","citing_title":"Token-Level Density-Based Uncertainty Quantification Methods for Eliciting Truthfulness of Large Language Models","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2503.18562","citing_title":"Self-Reported Confidence of Large Language Models in Gastroenterology: Analysis of Commercial, Open-Source, and Quantized Models","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2511.20284","citing_title":"Can LLMs Make (Personalized) Access Control Decisions?","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08346","citing_title":"Sanity Checks for Long-Form Hallucination Detection","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05777","citing_title":"Estimating the Black-box LLM Uncertainty with Distribution-Aligned Adversarial Distillation","ref_index":56,"is_internal_anchor":false},{"citing_arxiv_id":"2605.04295","citing_title":"LLMs Uncertainty Quantification via Adaptive Conformal Semantic Entropy","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2604.07825","citing_title":"Filling the Gaps: Selective Knowledge Augmentation for LLM Recommenders","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07353","citing_title":"Confidence-Aware Alignment Makes Reasoning LLMs More Reliable","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2604.15109","citing_title":"IUQ: Interrogative Uncertainty Quantification for Long-Form Large Language Model Generation","ref_index":12,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/LEKBR5LZ3IP3KI7Z6SYAIUSUFX","json":"https://pith.science/pith/LEKBR5LZ3IP3KI7Z6SYAIUSUFX.json","graph_json":"https://pith.science/api/pith-number/LEKBR5LZ3IP3KI7Z6SYAIUSUFX/graph.json","events_json":"https://pith.science/api/pith-number/LEKBR5LZ3IP3KI7Z6SYAIUSUFX/events.json","paper":"https://pith.science/paper/LEKBR5LZ"},"agent_actions":{"view_html":"https://pith.science/pith/LEKBR5LZ3IP3KI7Z6SYAIUSUFX","download_json":"https://pith.science/pith/LEKBR5LZ3IP3KI7Z6SYAIUSUFX.json","view_paper":"https://pith.science/paper/LEKBR5LZ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2403.04696&json=true","fetch_graph":"https://pith.science/api/pith-number/LEKBR5LZ3IP3KI7Z6SYAIUSUFX/graph.json","fetch_events":"https://pith.science/api/pith-number/LEKBR5LZ3IP3KI7Z6SYAIUSUFX/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/LEKBR5LZ3IP3KI7Z6SYAIUSUFX/action/timestamp_anchor","attest_storage":"https://pith.science/pith/LEKBR5LZ3IP3KI7Z6SYAIUSUFX/action/storage_attestation","attest_author":"https://pith.science/pith/LEKBR5LZ3IP3KI7Z6SYAIUSUFX/action/author_attestation","sign_citation":"https://pith.science/pith/LEKBR5LZ3IP3KI7Z6SYAIUSUFX/action/citation_signature","submit_replication":"https://pith.science/pith/LEKBR5LZ3IP3KI7Z6SYAIUSUFX/action/replication_record"}},"created_at":"2026-07-05T08:28:39.364190+00:00","updated_at":"2026-07-05T08:28:39.364190+00:00"}