{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:VTKBNIVMUHWIPFOPAWDEX37XX3","short_pith_number":"pith:VTKBNIVM","schema_version":"1.0","canonical_sha256":"acd416a2aca1ec8795cf05864beff7bec5161755eb9788614d80b3529aff51af","source":{"kind":"arxiv","id":"2306.13781","version":1},"attestation_state":"computed","paper":{"title":"Retrieving Supporting Evidence for LLMs Generated Answers","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.IR","authors_text":"Charles L. A. Clarke, Negar Arabzadeh, Siqing Huo","submitted_at":"2023-06-23T20:45:29Z","abstract_excerpt":"Current large language models (LLMs) can exhibit near-human levels of performance on many natural language tasks, including open-domain question answering. Unfortunately, they also convincingly hallucinate incorrect answers, so that responses to questions must be verified against external sources before they can be accepted at face value. In this paper, we report a simple experiment to automatically verify generated answers against a corpus. After presenting a question to an LLM and receiving a generated answer, we query the corpus with the combination of the question + generated answer. We th"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2306.13781","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.IR","submitted_at":"2023-06-23T20:45:29Z","cross_cats_sorted":[],"title_canon_sha256":"402256d80b60b41e7fd2674273947e3bcf17a2d2bb7367dfc7648e4822bbadf1","abstract_canon_sha256":"4b22bb2152078b21b170f7acc18bd8ee507cd9b6ba1060add886dc4e0372871a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:24:15.612071Z","signature_b64":"7MZcxDMh5+iKuKDWg4HV6yxZFfTPxSM11DiRl1xes4KbttD0DdAtnzfHqtIFJV5DtojmR/rbCXmMHmHOvUNbBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"acd416a2aca1ec8795cf05864beff7bec5161755eb9788614d80b3529aff51af","last_reissued_at":"2026-07-05T06:24:15.611621Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:24:15.611621Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Retrieving Supporting Evidence for LLMs Generated Answers","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.IR","authors_text":"Charles L. A. Clarke, Negar Arabzadeh, Siqing Huo","submitted_at":"2023-06-23T20:45:29Z","abstract_excerpt":"Current large language models (LLMs) can exhibit near-human levels of performance on many natural language tasks, including open-domain question answering. Unfortunately, they also convincingly hallucinate incorrect answers, so that responses to questions must be verified against external sources before they can be accepted at face value. In this paper, we report a simple experiment to automatically verify generated answers against a corpus. After presenting a question to an LLM and receiving a generated answer, we query the corpus with the combination of the question + generated answer. We th"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2306.13781","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2306.13781/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2306.13781","created_at":"2026-07-05T06:24:15.611679+00:00"},{"alias_kind":"arxiv_version","alias_value":"2306.13781v1","created_at":"2026-07-05T06:24:15.611679+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2306.13781","created_at":"2026-07-05T06:24:15.611679+00:00"},{"alias_kind":"pith_short_12","alias_value":"VTKBNIVMUHWI","created_at":"2026-07-05T06:24:15.611679+00:00"},{"alias_kind":"pith_short_16","alias_value":"VTKBNIVMUHWIPFOP","created_at":"2026-07-05T06:24:15.611679+00:00"},{"alias_kind":"pith_short_8","alias_value":"VTKBNIVM","created_at":"2026-07-05T06:24:15.611679+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2508.18473","citing_title":"Principled Detection of Hallucinations in Large Language Models via Multiple Testing","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2311.05232","citing_title":"A Survey on Hallucination in Large Language Models: Principles, Taxonomy, Challenges, and Open Questions","ref_index":133,"is_internal_anchor":false},{"citing_arxiv_id":"2605.04180","citing_title":"MedFabric and EtHER: A Data-Centric Framework for Word-Level Fabrication Generation and Detection in Medical LLMs","ref_index":10,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/VTKBNIVMUHWIPFOPAWDEX37XX3","json":"https://pith.science/pith/VTKBNIVMUHWIPFOPAWDEX37XX3.json","graph_json":"https://pith.science/api/pith-number/VTKBNIVMUHWIPFOPAWDEX37XX3/graph.json","events_json":"https://pith.science/api/pith-number/VTKBNIVMUHWIPFOPAWDEX37XX3/events.json","paper":"https://pith.science/paper/VTKBNIVM"},"agent_actions":{"view_html":"https://pith.science/pith/VTKBNIVMUHWIPFOPAWDEX37XX3","download_json":"https://pith.science/pith/VTKBNIVMUHWIPFOPAWDEX37XX3.json","view_paper":"https://pith.science/paper/VTKBNIVM","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2306.13781&json=true","fetch_graph":"https://pith.science/api/pith-number/VTKBNIVMUHWIPFOPAWDEX37XX3/graph.json","fetch_events":"https://pith.science/api/pith-number/VTKBNIVMUHWIPFOPAWDEX37XX3/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/VTKBNIVMUHWIPFOPAWDEX37XX3/action/timestamp_anchor","attest_storage":"https://pith.science/pith/VTKBNIVMUHWIPFOPAWDEX37XX3/action/storage_attestation","attest_author":"https://pith.science/pith/VTKBNIVMUHWIPFOPAWDEX37XX3/action/author_attestation","sign_citation":"https://pith.science/pith/VTKBNIVMUHWIPFOPAWDEX37XX3/action/citation_signature","submit_replication":"https://pith.science/pith/VTKBNIVMUHWIPFOPAWDEX37XX3/action/replication_record"}},"created_at":"2026-07-05T06:24:15.611679+00:00","updated_at":"2026-07-05T06:24:15.611679+00:00"}