{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:PZMHP7F6BS22T5RIU3WQYBDRIL","short_pith_number":"pith:PZMHP7F6","schema_version":"1.0","canonical_sha256":"7e5877fcbe0cb5a9f628a6ed0c047142cd8c2f90658a303daba4da98ac33aa42","source":{"kind":"arxiv","id":"2305.11747","version":3},"attestation_state":"computed","paper":{"title":"HaluEval: A Large-Scale Hallucination Evaluation Benchmark for Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Jian-Yun Nie, Ji-Rong Wen, Junyi Li, Wayne Xin Zhao, Xiaoxue Cheng","submitted_at":"2023-05-19T15:36:27Z","abstract_excerpt":"Large language models (LLMs), such as ChatGPT, are prone to generate hallucinations, i.e., content that conflicts with the source or cannot be verified by the factual knowledge. To understand what types of content and to which extent LLMs are apt to hallucinate, we introduce the Hallucination Evaluation benchmark for Large Language Models (HaluEval), a large collection of generated and human-annotated hallucinated samples for evaluating the performance of LLMs in recognizing hallucination. To generate these samples, we propose a ChatGPT-based two-step framework, i.e., sampling-then-filtering. "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2305.11747","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2023-05-19T15:36:27Z","cross_cats_sorted":[],"title_canon_sha256":"02c763aec9d5064d1eb7c18dc6d5ef2353725ebd3ac15a51fe90a68e654d8580","abstract_canon_sha256":"0a44b785a21811afc6c3ac560b93ae1076151577b2314d7d166d58ac438d68d5"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:03:24.877642Z","signature_b64":"o/gZVjBatFb7BFtj8Zjz6hqSx5ehsbsW1KJJcNQGRUJUo6lMn3wTQh46GFaJTBGwNUGYtVGaFjZSJyPw85LiDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"7e5877fcbe0cb5a9f628a6ed0c047142cd8c2f90658a303daba4da98ac33aa42","last_reissued_at":"2026-07-05T07:03:24.877151Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:03:24.877151Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"HaluEval: A Large-Scale Hallucination Evaluation Benchmark for Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Jian-Yun Nie, Ji-Rong Wen, Junyi Li, Wayne Xin Zhao, Xiaoxue Cheng","submitted_at":"2023-05-19T15:36:27Z","abstract_excerpt":"Large language models (LLMs), such as ChatGPT, are prone to generate hallucinations, i.e., content that conflicts with the source or cannot be verified by the factual knowledge. To understand what types of content and to which extent LLMs are apt to hallucinate, we introduce the Hallucination Evaluation benchmark for Large Language Models (HaluEval), a large collection of generated and human-annotated hallucinated samples for evaluating the performance of LLMs in recognizing hallucination. To generate these samples, we propose a ChatGPT-based two-step framework, i.e., sampling-then-filtering. "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2305.11747","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2305.11747/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2305.11747","created_at":"2026-07-05T07:03:24.877209+00:00"},{"alias_kind":"arxiv_version","alias_value":"2305.11747v3","created_at":"2026-07-05T07:03:24.877209+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2305.11747","created_at":"2026-07-05T07:03:24.877209+00:00"},{"alias_kind":"pith_short_12","alias_value":"PZMHP7F6BS22","created_at":"2026-07-05T07:03:24.877209+00:00"},{"alias_kind":"pith_short_16","alias_value":"PZMHP7F6BS22T5RI","created_at":"2026-07-05T07:03:24.877209+00:00"},{"alias_kind":"pith_short_8","alias_value":"PZMHP7F6","created_at":"2026-07-05T07:03:24.877209+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":24,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.13220","citing_title":"LLM-as-an-Investigator: Evidence-First Reasoning for Robust Interactive Problem Diagnosis","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2606.08705","citing_title":"Analyzing the Correlation Between Hallucinations and Knowledge Conflicts in Large Language Models","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2606.04435","citing_title":"Cascading Hallucination in Agentic RAG: The CHARM Framework for Detection and Mitigation","ref_index":38,"is_internal_anchor":false},{"citing_arxiv_id":"2606.32032","citing_title":"Reinforcement Learning with Metacognitive Feedback Elicits Faithful Uncertainty Expression in LLMs","ref_index":59,"is_internal_anchor":false},{"citing_arxiv_id":"2605.24919","citing_title":"MultiHaluDet: Multilingual Hallucination Detection via LLM Hidden State Probing","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2605.28778","citing_title":"Can LLMs Use Linguistic Uncertainty Markers to Reliably Reflect Intrinsic Confidence?","ref_index":49,"is_internal_anchor":false},{"citing_arxiv_id":"2605.29523","citing_title":"K-FinHallu: A Hallucination Detection Benchmark for Multi-Turn RAG in Korean Finance","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2605.02443","citing_title":"HalluScan: A Systematic Benchmark for Detecting and Mitigating Hallucinations in Instruction-Following LLMs","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2605.23262","citing_title":"Design and Report Benchmarks for Knowledge Work","ref_index":61,"is_internal_anchor":false},{"citing_arxiv_id":"2504.00446","citing_title":"Exposing the Ghost in the Transformer: Abnormal Detection for Large Language Models via Hidden State Forensics","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2603.14987","citing_title":"Beyond Benchmark Islands: Toward Representative Trustworthiness Evaluation for Agentic AI","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2510.07239","citing_title":"Red-Bandit: Test-Time Adaptation for LLM Red-Teaming via Bandit-Guided LoRA Experts","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17007","citing_title":"HalluScore: Large Language Model Hallucination Question Answering Benchmark","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2309.15217","citing_title":"Ragas: Automated Evaluation of Retrieval Augmented Generation","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2601.03846","citing_title":"When Numbers Start Talking: Implicit Numerical Coordination Among LLM-Based Agents","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2309.05922","citing_title":"A Survey of Hallucination in Large Foundation Models","ref_index":129,"is_internal_anchor":false},{"citing_arxiv_id":"2602.06718","citing_title":"GhostCite: A Large-Scale Analysis of Citation Validity in the Age of Large Language Models","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2411.04368","citing_title":"Measuring short-form factuality in large language models","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05810","citing_title":"CXR-ContraBench: Benchmarking Negated-Option Attraction in Medical VLMs","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2604.10734","citing_title":"Self-Correcting RAG: Enhancing Faithfulness via MMKP Context Selection and NLI-Guided MCTS","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2604.10733","citing_title":"Too Nice to Tell the Truth: Quantifying Agreeableness-Driven Sycophancy in Role-Playing Language Models","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2604.08519","citing_title":"Cram Less to Fit More: Training Data Pruning Improves Memorization of Facts","ref_index":50,"is_internal_anchor":false},{"citing_arxiv_id":"2604.04743","citing_title":"Hallucination Basins: A Dynamic Framework for Understanding and Controlling LLM Hallucinations","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2604.15945","citing_title":"RAGognizer: Hallucination-Aware Fine-Tuning via Detection Head Integration","ref_index":28,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/PZMHP7F6BS22T5RIU3WQYBDRIL","json":"https://pith.science/pith/PZMHP7F6BS22T5RIU3WQYBDRIL.json","graph_json":"https://pith.science/api/pith-number/PZMHP7F6BS22T5RIU3WQYBDRIL/graph.json","events_json":"https://pith.science/api/pith-number/PZMHP7F6BS22T5RIU3WQYBDRIL/events.json","paper":"https://pith.science/paper/PZMHP7F6"},"agent_actions":{"view_html":"https://pith.science/pith/PZMHP7F6BS22T5RIU3WQYBDRIL","download_json":"https://pith.science/pith/PZMHP7F6BS22T5RIU3WQYBDRIL.json","view_paper":"https://pith.science/paper/PZMHP7F6","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2305.11747&json=true","fetch_graph":"https://pith.science/api/pith-number/PZMHP7F6BS22T5RIU3WQYBDRIL/graph.json","fetch_events":"https://pith.science/api/pith-number/PZMHP7F6BS22T5RIU3WQYBDRIL/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/PZMHP7F6BS22T5RIU3WQYBDRIL/action/timestamp_anchor","attest_storage":"https://pith.science/pith/PZMHP7F6BS22T5RIU3WQYBDRIL/action/storage_attestation","attest_author":"https://pith.science/pith/PZMHP7F6BS22T5RIU3WQYBDRIL/action/author_attestation","sign_citation":"https://pith.science/pith/PZMHP7F6BS22T5RIU3WQYBDRIL/action/citation_signature","submit_replication":"https://pith.science/pith/PZMHP7F6BS22T5RIU3WQYBDRIL/action/replication_record"}},"created_at":"2026-07-05T07:03:24.877209+00:00","updated_at":"2026-07-05T07:03:24.877209+00:00"}