{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:AZKGABXCOM3YGUNDADAYYLZ35H","short_pith_number":"pith:AZKGABXC","schema_version":"1.0","canonical_sha256":"06546006e273378351a300c18c2f3be9f1ad640908645227e1590f0707718f3a","source":{"kind":"arxiv","id":"2311.15296","version":3},"attestation_state":"computed","paper":{"title":"UHGEval: Benchmarking the Hallucination of Chinese Large Language Models via Unconstrained Generation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Bo Tang, Dawei He, Feiyu Xiong, Haiying Deng, Peng Cheng, Shichao Song, Simin Niu, Xun Liang, Yezhaohui Wang, Zhiyu Li, Zhonghao Wang","submitted_at":"2023-11-26T13:42:56Z","abstract_excerpt":"Large language models (LLMs) have emerged as pivotal contributors in contemporary natural language processing and are increasingly being applied across a diverse range of industries. However, these large-scale probabilistic statistical models cannot currently ensure the requisite quality in professional content generation. These models often produce hallucinated text, compromising their practical utility in professional contexts. To assess the authentic reliability of LLMs in text generation, numerous initiatives have developed benchmark evaluations for hallucination phenomena. Nevertheless, t"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2311.15296","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2023-11-26T13:42:56Z","cross_cats_sorted":[],"title_canon_sha256":"1c9cd24593e5384938ead91fd0e322a7159137f8f07c3b88cd454bd78bf8aaf3","abstract_canon_sha256":"ebc17635e3573ff88ba71566f8d53d06b55e38a468b7dcfd5956aa885d840502"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:17:34.967986Z","signature_b64":"Jd0p+kuP6tXG+jgN3IJNT4Xq9Q6QdCq4VQDAam/duTrLlGv41CxDYuiYhteG4dVG+8Oosxjt/NUtKm47Qq/7Cg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"06546006e273378351a300c18c2f3be9f1ad640908645227e1590f0707718f3a","last_reissued_at":"2026-07-05T09:17:34.967459Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:17:34.967459Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"UHGEval: Benchmarking the Hallucination of Chinese Large Language Models via Unconstrained Generation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Bo Tang, Dawei He, Feiyu Xiong, Haiying Deng, Peng Cheng, Shichao Song, Simin Niu, Xun Liang, Yezhaohui Wang, Zhiyu Li, Zhonghao Wang","submitted_at":"2023-11-26T13:42:56Z","abstract_excerpt":"Large language models (LLMs) have emerged as pivotal contributors in contemporary natural language processing and are increasingly being applied across a diverse range of industries. However, these large-scale probabilistic statistical models cannot currently ensure the requisite quality in professional content generation. These models often produce hallucinated text, compromising their practical utility in professional contexts. To assess the authentic reliability of LLMs in text generation, numerous initiatives have developed benchmark evaluations for hallucination phenomena. Nevertheless, t"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2311.15296","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2311.15296/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2311.15296","created_at":"2026-07-05T09:17:34.967529+00:00"},{"alias_kind":"arxiv_version","alias_value":"2311.15296v3","created_at":"2026-07-05T09:17:34.967529+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2311.15296","created_at":"2026-07-05T09:17:34.967529+00:00"},{"alias_kind":"pith_short_12","alias_value":"AZKGABXCOM3Y","created_at":"2026-07-05T09:17:34.967529+00:00"},{"alias_kind":"pith_short_16","alias_value":"AZKGABXCOM3YGUND","created_at":"2026-07-05T09:17:34.967529+00:00"},{"alias_kind":"pith_short_8","alias_value":"AZKGABXC","created_at":"2026-07-05T09:17:34.967529+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2506.18421","citing_title":"TReB: A Comprehensive Benchmark for Evaluating Table Reasoning Capabilities of Large Language Models","ref_index":22,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/AZKGABXCOM3YGUNDADAYYLZ35H","json":"https://pith.science/pith/AZKGABXCOM3YGUNDADAYYLZ35H.json","graph_json":"https://pith.science/api/pith-number/AZKGABXCOM3YGUNDADAYYLZ35H/graph.json","events_json":"https://pith.science/api/pith-number/AZKGABXCOM3YGUNDADAYYLZ35H/events.json","paper":"https://pith.science/paper/AZKGABXC"},"agent_actions":{"view_html":"https://pith.science/pith/AZKGABXCOM3YGUNDADAYYLZ35H","download_json":"https://pith.science/pith/AZKGABXCOM3YGUNDADAYYLZ35H.json","view_paper":"https://pith.science/paper/AZKGABXC","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2311.15296&json=true","fetch_graph":"https://pith.science/api/pith-number/AZKGABXCOM3YGUNDADAYYLZ35H/graph.json","fetch_events":"https://pith.science/api/pith-number/AZKGABXCOM3YGUNDADAYYLZ35H/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/AZKGABXCOM3YGUNDADAYYLZ35H/action/timestamp_anchor","attest_storage":"https://pith.science/pith/AZKGABXCOM3YGUNDADAYYLZ35H/action/storage_attestation","attest_author":"https://pith.science/pith/AZKGABXCOM3YGUNDADAYYLZ35H/action/author_attestation","sign_citation":"https://pith.science/pith/AZKGABXCOM3YGUNDADAYYLZ35H/action/citation_signature","submit_replication":"https://pith.science/pith/AZKGABXCOM3YGUNDADAYYLZ35H/action/replication_record"}},"created_at":"2026-07-05T09:17:34.967529+00:00","updated_at":"2026-07-05T09:17:34.967529+00:00"}