{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:FX7ZWR4QG65M46BZTRB63EWE4G","short_pith_number":"pith:FX7ZWR4Q","schema_version":"1.0","canonical_sha256":"2dff9b479037bace78399c43ed92c4e1b71093663d840e338f91f7015ddc7d1f","source":{"kind":"arxiv","id":"2401.00396","version":2},"attestation_state":"computed","paper":{"title":"RAGTruth: A Hallucination Corpus for Developing Trustworthy Retrieval-Augmented Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Cheng Niu, Juno Zhu, Juntong Song, Kashun Shum, Randy Zhong, Siliang Xu, Tong Zhang, Yuanhao Wu","submitted_at":"2023-12-31T04:43:45Z","abstract_excerpt":"Retrieval-augmented generation (RAG) has become a main technique for alleviating hallucinations in large language models (LLMs). Despite the integration of RAG, LLMs may still present unsupported or contradictory claims to the retrieved contents. In order to develop effective hallucination prevention strategies under RAG, it is important to create benchmark datasets that can measure the extent of hallucination. This paper presents RAGTruth, a corpus tailored for analyzing word-level hallucinations in various domains and tasks within the standard RAG frameworks for LLM applications. RAGTruth co"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2401.00396","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2023-12-31T04:43:45Z","cross_cats_sorted":[],"title_canon_sha256":"538e1209a9b1ccb0b459feb90dfc86b7f1558045790fe123cf09ef074d8649f7","abstract_canon_sha256":"dbf6d3e8b3736b7cbd64c74603d6a64cdc29d85cdbf659d7354c5637369037b8"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:20:09.310890Z","signature_b64":"rlczz748LnCK6g/5/ZgP1twhkIZjajdD1ngBlETS3lPBe4pwcsM2cUMP7LzDyIw/4JbWq1fFN7H8h42EUlGjAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"2dff9b479037bace78399c43ed92c4e1b71093663d840e338f91f7015ddc7d1f","last_reissued_at":"2026-07-05T08:20:09.310418Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:20:09.310418Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"RAGTruth: A Hallucination Corpus for Developing Trustworthy Retrieval-Augmented Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Cheng Niu, Juno Zhu, Juntong Song, Kashun Shum, Randy Zhong, Siliang Xu, Tong Zhang, Yuanhao Wu","submitted_at":"2023-12-31T04:43:45Z","abstract_excerpt":"Retrieval-augmented generation (RAG) has become a main technique for alleviating hallucinations in large language models (LLMs). Despite the integration of RAG, LLMs may still present unsupported or contradictory claims to the retrieved contents. In order to develop effective hallucination prevention strategies under RAG, it is important to create benchmark datasets that can measure the extent of hallucination. This paper presents RAGTruth, a corpus tailored for analyzing word-level hallucinations in various domains and tasks within the standard RAG frameworks for LLM applications. RAGTruth co"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2401.00396","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2401.00396/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2401.00396","created_at":"2026-07-05T08:20:09.310475+00:00"},{"alias_kind":"arxiv_version","alias_value":"2401.00396v2","created_at":"2026-07-05T08:20:09.310475+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2401.00396","created_at":"2026-07-05T08:20:09.310475+00:00"},{"alias_kind":"pith_short_12","alias_value":"FX7ZWR4QG65M","created_at":"2026-07-05T08:20:09.310475+00:00"},{"alias_kind":"pith_short_16","alias_value":"FX7ZWR4QG65M46BZ","created_at":"2026-07-05T08:20:09.310475+00:00"},{"alias_kind":"pith_short_8","alias_value":"FX7ZWR4Q","created_at":"2026-07-05T08:20:09.310475+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":11,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.26437","citing_title":"ConflictScore: Identifying and Measuring How Language Models Handle Conflicting Evidence","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2606.30306","citing_title":"Always-OnAgents:A Survey of Persistent Memory, State, and Governance in LLMAgents","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2605.26778","citing_title":"The Attribution Blind Spot: Detecting When Language Models Rely on Memory Rather Than Retrieved Context","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2605.27494","citing_title":"Grounded Cache Routing for Retrieval-Augmented Generation: When Is It Safe to Reuse an Answer?","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2504.10063","citing_title":"Hallucination Detection in LLMs with Topological Divergence on Attention Graphs","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2509.07794","citing_title":"Query Expansion in the Age of Pre-trained and Large Language Models: A Comprehensive Survey","ref_index":79,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12813","citing_title":"REALISTA: Realistic Latent Adversarial Attacks that Elicit LLM Hallucinations","ref_index":176,"is_internal_anchor":false},{"citing_arxiv_id":"2604.09493","citing_title":"Policy-Aware Edge LLM-RAG Framework for Internet of Battlefield Things Mission Orchestration","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2604.15945","citing_title":"RAGognizer: Hallucination-Aware Fine-Tuning via Detection Head Integration","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2604.16706","citing_title":"Auditing Automated Evaluation, Error Propagation, and Runtime Mitigation in Tool-Using Language Agents","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2605.03140","citing_title":"Evaluating Retrieval-Augmented Generation for Explainable Malware Analysis","ref_index":21,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/FX7ZWR4QG65M46BZTRB63EWE4G","json":"https://pith.science/pith/FX7ZWR4QG65M46BZTRB63EWE4G.json","graph_json":"https://pith.science/api/pith-number/FX7ZWR4QG65M46BZTRB63EWE4G/graph.json","events_json":"https://pith.science/api/pith-number/FX7ZWR4QG65M46BZTRB63EWE4G/events.json","paper":"https://pith.science/paper/FX7ZWR4Q"},"agent_actions":{"view_html":"https://pith.science/pith/FX7ZWR4QG65M46BZTRB63EWE4G","download_json":"https://pith.science/pith/FX7ZWR4QG65M46BZTRB63EWE4G.json","view_paper":"https://pith.science/paper/FX7ZWR4Q","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2401.00396&json=true","fetch_graph":"https://pith.science/api/pith-number/FX7ZWR4QG65M46BZTRB63EWE4G/graph.json","fetch_events":"https://pith.science/api/pith-number/FX7ZWR4QG65M46BZTRB63EWE4G/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/FX7ZWR4QG65M46BZTRB63EWE4G/action/timestamp_anchor","attest_storage":"https://pith.science/pith/FX7ZWR4QG65M46BZTRB63EWE4G/action/storage_attestation","attest_author":"https://pith.science/pith/FX7ZWR4QG65M46BZTRB63EWE4G/action/author_attestation","sign_citation":"https://pith.science/pith/FX7ZWR4QG65M46BZTRB63EWE4G/action/citation_signature","submit_replication":"https://pith.science/pith/FX7ZWR4QG65M46BZTRB63EWE4G/action/replication_record"}},"created_at":"2026-07-05T08:20:09.310475+00:00","updated_at":"2026-07-05T08:20:09.310475+00:00"}