{"as_of":"2026-08-09T14:56:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:653885fa65b19d8d500f81c4a0ab0a22458f440f30d6aa8b4fb6e536e3a0f748","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":52,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":52,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-09T06:31:02.800959+00:00","state":"measured"},{"denominator":52,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":52,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-08T23:35:38.467302Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":1,"source":"arxiv_reference","source_observed_at":"2026-08-05T02:28:24.338817Z","state":"measured"}],"external_citation_measurements":[{"count":33,"observed_at":"2026-08-05T02:28:24.338817Z","source":"arxiv_reference"}],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2305.11747","last_updated":"2023-10-23T01:49:32Z","snapshot_observed_at":"2026-08-07T22:56:15.588770Z","submitted_at":"2023-05-19T15:36:27Z","title":"HaluEval: A Large-Scale Hallucination Evaluation Benchmark for Large Language Models","version":3},"cited_work":{"arxiv_id":"2305.11747","doi":"10.48550/arxiv.2305.11747","metadata_source":"arxiv_reference","pith_arxiv_id":"2305.11747","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv preprint arXiv:2305.11747 (2023)","venue":"arXiv (Cornell University)","work_id":"2cf6bc2d-aed3-4a58-879e-daf0de687940","year":2023},"citing_paper":{"arxiv_id":"2309.05922","last_updated":"2023-09-12T02:34:06Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-09-12T02:34:06Z","title":"A Survey of Hallucination in Large Foundation Models","version":1},"reference_index":129,"source":"arxiv_source","source_observed_at":"2026-05-16T15:21:00.778049Z"},"links":{"cited_paper":"/paper/2305.11747","citing_paper":"/paper/2309.05922"},"observation_digest":"sha256:b7e15039d6609aa2efb1548d4d71de1d769c75fd806fe6bff6f74df658116891","observation_id":"f785cd9a-dd14-4950-938f-38ac44edc2ae","resolution":{"observed_at":"2026-05-16T15:21:00.927160Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.11747","last_updated":"2023-10-23T01:49:32Z","snapshot_observed_at":"2026-08-07T22:56:15.588770Z","submitted_at":"2023-05-19T15:36:27Z","title":"HaluEval: A Large-Scale Hallucination Evaluation Benchmark for Large Language Models","version":3},"cited_work":{"arxiv_id":"2305.11747","doi":"10.48550/arxiv.2305.11747","metadata_source":"arxiv_reference","pith_arxiv_id":"2305.11747","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv preprint arXiv:2305.11747 (2023)","venue":"arXiv (Cornell University)","work_id":"2cf6bc2d-aed3-4a58-879e-daf0de687940","year":2023},"citing_paper":{"arxiv_id":"2309.15217","last_updated":"2025-04-28T05:09:12Z","snapshot_observed_at":"2026-08-05T02:02:34.574070Z","submitted_at":"2023-09-26T19:23:54Z","title":"Ragas: Automated Evaluation of Retrieval Augmented Generation","version":2},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-16T21:37:40.907162Z"},"links":{"cited_paper":"/paper/2305.11747","citing_paper":"/paper/2309.15217"},"observation_digest":"sha256:746a89bfc4b1a788ee68095face8fed8ffc7120ed8d5f273dfc2d804780df9d6","observation_id":"3ad50d03-cfc5-42ae-8f07-8c17fd598d2e","resolution":{"observed_at":"2026-05-16T21:37:40.921250Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.11747","last_updated":"2023-10-23T01:49:32Z","snapshot_observed_at":"2026-08-07T22:56:15.588770Z","submitted_at":"2023-05-19T15:36:27Z","title":"HaluEval: A Large-Scale Hallucination Evaluation Benchmark for Large Language Models","version":3},"cited_work":{"arxiv_id":"2305.11747","doi":"10.48550/arxiv.2305.11747","metadata_source":"arxiv_reference","pith_arxiv_id":"2305.11747","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv preprint arXiv:2305.11747 (2023)","venue":"arXiv (Cornell University)","work_id":"2cf6bc2d-aed3-4a58-879e-daf0de687940","year":2023},"citing_paper":{"arxiv_id":"2411.04368","last_updated":"2024-11-07T01:58:42Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-11-07T01:58:42Z","title":"Measuring short-form factuality in large language models","version":1},"reference_index":9,"source":"arxiv_source","source_observed_at":"2026-05-15T06:45:50.219157Z"},"links":{"cited_paper":"/paper/2305.11747","citing_paper":"/paper/2411.04368"},"observation_digest":"sha256:e3a51b33149195851c2b4d26c3ccbee4845b372be4c854fd165d57f02cde1e3d","observation_id":"d3b46f2b-f045-40d8-9e11-be3075ccf31c","resolution":{"observed_at":"2026-05-15T06:45:50.262718Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.11747","last_updated":"2023-10-23T01:49:32Z","snapshot_observed_at":"2026-08-07T22:56:15.588770Z","submitted_at":"2023-05-19T15:36:27Z","title":"HaluEval: A Large-Scale Hallucination Evaluation Benchmark for Large Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.11747","snapshot_observed_at":"2026-08-08T23:35:38.467302Z","title":"Halueval: A large-scale hallucination evaluation benchmark for large language models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2502.04095","last_updated":"2025-02-06T14:12:41Z","snapshot_observed_at":"2026-08-08T23:30:20.714032Z","submitted_at":"2025-02-06T14:12:41Z","title":"LLMs to Support a Domain Specific Knowledge Assistant","version":1},"reference_index":41,"source":"arxiv_source","source_observed_at":"2026-08-08T23:35:38.467302Z"},"links":{"cited_paper":"/paper/2305.11747","citing_paper":"/paper/2502.04095"},"observation_digest":"sha256:540da350673c001a14034281c2bd18ed662378d6c44f7221a93e78f174525433","observation_id":"ba0a48c1-541e-4cba-ba66-c9c0066dbb80","resolution":{"observed_at":"2026-08-08T23:35:38.467302Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.11747","last_updated":"2023-10-23T01:49:32Z","snapshot_observed_at":"2026-08-07T22:56:15.588770Z","submitted_at":"2023-05-19T15:36:27Z","title":"HaluEval: A Large-Scale Hallucination Evaluation Benchmark for Large Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.11747","snapshot_observed_at":"2026-08-08T22:23:28.190876Z","title":"X., Nie, J.-Y., and Wen, J.-R","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2502.04556","last_updated":"2025-02-06T23:10:14Z","snapshot_observed_at":"2026-08-09T04:46:44.669056Z","submitted_at":"2025-02-06T23:10:14Z","title":"TruthFlow: Truthful LLM Generation via Representation Flow Correction","version":1},"reference_index":31,"source":"arxiv_source","source_observed_at":"2026-08-08T22:23:28.190876Z"},"links":{"cited_paper":"/paper/2305.11747","citing_paper":"/paper/2502.04556"},"observation_digest":"sha256:36f55b9325e81bfadc230e2fe78139357ef2f4d0cd7c43b4b8de55d5c391775f","observation_id":"ba465a51-296c-4227-881a-eef3ca3c7eed","resolution":{"observed_at":"2026-08-08T22:23:28.190876Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.11747","last_updated":"2023-10-23T01:49:32Z","snapshot_observed_at":"2026-08-07T22:56:15.588770Z","submitted_at":"2023-05-19T15:36:27Z","title":"HaluEval: A Large-Scale Hallucination Evaluation Benchmark for Large Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.11747","snapshot_observed_at":"2026-08-08T18:23:05.799001Z","title":"Halueval: A large-scale hallucination evaluation benchmark for large language models,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2502.06884","last_updated":"2025-02-08T21:30:41Z","snapshot_observed_at":"2026-08-08T18:16:07.634775Z","submitted_at":"2025-02-08T21:30:41Z","title":"Learning Conformal Abstention Policies for Adaptive Risk Management in Large Language and Vision-Language Models","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-08T18:23:05.799001Z"},"links":{"cited_paper":"/paper/2305.11747","citing_paper":"/paper/2502.06884"},"observation_digest":"sha256:026cbe4df998319acfee04834c8161a184194c2a0ceeef08b14d06715acecfff","observation_id":"aa4d4467-2c3e-4e9d-851e-8e2b8c873dd6","resolution":{"observed_at":"2026-08-08T18:23:05.799001Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.11747","last_updated":"2023-10-23T01:49:32Z","snapshot_observed_at":"2026-08-07T22:56:15.588770Z","submitted_at":"2023-05-19T15:36:27Z","title":"HaluEval: A Large-Scale Hallucination Evaluation Benchmark for Large Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.11747","snapshot_observed_at":"2026-08-07T23:19:36.717618Z","title":"arXiv preprint arXiv:2305.11747","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2502.08904","last_updated":"2025-02-27T01:49:15Z","snapshot_observed_at":"2026-08-08T16:15:05.187445Z","submitted_at":"2025-02-13T02:40:33Z","title":"MIH-TCCT: Mitigating Inconsistent Hallucinations in LLMs via Event-Driven Text-Code Cyclic Training","version":3},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-07T23:19:36.717618Z"},"links":{"cited_paper":"/paper/2305.11747","citing_paper":"/paper/2502.08904"},"observation_digest":"sha256:6addb7e356d90e651c6272aaf90c61422b1cbfaa7036bd2b66e8c0a03ad219a6","observation_id":"9e01f88b-2345-4b57-9cc3-a5da29ba029f","resolution":{"observed_at":"2026-08-07T23:19:36.717618Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.11747","last_updated":"2023-10-23T01:49:32Z","snapshot_observed_at":"2026-08-07T22:56:15.588770Z","submitted_at":"2023-05-19T15:36:27Z","title":"HaluEval: A Large-Scale Hallucination Evaluation Benchmark for Large Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.11747","snapshot_observed_at":"2026-08-07T23:35:42.740185Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2502.09670","last_updated":"2025-02-12T22:55:43Z","snapshot_observed_at":"2026-08-09T08:58:06.881734Z","submitted_at":"2025-02-12T22:55:43Z","title":"The Science of Evaluating Foundation Models","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-07T23:35:42.740185Z"},"links":{"cited_paper":"/paper/2305.11747","citing_paper":"/paper/2502.09670"},"observation_digest":"sha256:506aecde731249bb7665e7330687dcf75998323b59d3024584abb043f5332b87","observation_id":"90b11f8a-52fb-4b25-9266-8b7ae0f47d75","resolution":{"observed_at":"2026-08-07T23:35:42.740185Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.11747","last_updated":"2023-10-23T01:49:32Z","snapshot_observed_at":"2026-08-07T22:56:15.588770Z","submitted_at":"2023-05-19T15:36:27Z","title":"HaluEval: A Large-Scale Hallucination Evaluation Benchmark for Large Language Models","version":3},"cited_work":{"arxiv_id":"2305.11747","doi":"10.48550/arxiv.2305.11747","metadata_source":"arxiv_reference","pith_arxiv_id":"2305.11747","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv preprint arXiv:2305.11747 (2023)","venue":"arXiv (Cornell University)","work_id":"2cf6bc2d-aed3-4a58-879e-daf0de687940","year":2023},"citing_paper":{"arxiv_id":"2504.00446","last_updated":"2026-04-05T08:03:35Z","snapshot_observed_at":"2026-08-08T22:42:49.839017Z","submitted_at":"2025-04-01T05:58:14Z","title":"Exposing the Ghost in the Transformer: Abnormal Detection for Large Language Models via Hidden State Forensics","version":2},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-05-22T22:27:18.533162Z"},"links":{"cited_paper":"/paper/2305.11747","citing_paper":"/paper/2504.00446"},"observation_digest":"sha256:d2e34b55e3029b6e14dda9d029d9efaf945b624acc05022ef47f7d320d56b461","observation_id":"335fa9a4-8a13-4ac9-9e57-ff735608b911","resolution":{"observed_at":"2026-05-22T22:32:12.959996Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.11747","last_updated":"2023-10-23T01:49:32Z","snapshot_observed_at":"2026-08-07T22:56:15.588770Z","submitted_at":"2023-05-19T15:36:27Z","title":"HaluEval: A Large-Scale Hallucination Evaluation Benchmark for Large Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.11747","snapshot_observed_at":"2026-08-07T14:49:59.159168Z","title":"Halueval: A large-scale hallucination evaluation benchmark for large language models.arXiv preprint arXiv:2305.11747, 2023a","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.17529","last_updated":"2025-05-23T06:35:43Z","snapshot_observed_at":"2026-08-08T23:49:46.077669Z","submitted_at":"2025-05-23T06:35:43Z","title":"Do You Keep an Eye on What I Ask? Mitigating Multimodal Hallucination via Attention-Guided Ensemble Decoding","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-07T14:49:59.159168Z"},"links":{"cited_paper":"/paper/2305.11747","citing_paper":"/paper/2505.17529"},"observation_digest":"sha256:146c5d2c12c9e6b69bb424585b2bc123059310720ec6135be4061e0411a5ee33","observation_id":"64283d9c-668a-4ea5-a299-eefde29d0344","resolution":{"observed_at":"2026-08-07T14:49:59.159168Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.11747","last_updated":"2023-10-23T01:49:32Z","snapshot_observed_at":"2026-08-07T22:56:15.588770Z","submitted_at":"2023-05-19T15:36:27Z","title":"HaluEval: A Large-Scale Hallucination Evaluation Benchmark for Large Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.11747","snapshot_observed_at":"2026-08-07T14:50:38.384166Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.17558","last_updated":"2025-05-23T07:05:09Z","snapshot_observed_at":"2026-08-07T21:43:16.805139Z","submitted_at":"2025-05-23T07:05:09Z","title":"Teaching with Lies: Curriculum DPO on Synthetic Negatives for Hallucination Detection","version":1},"reference_index":16,"source":"arxiv_source","source_observed_at":"2026-08-07T14:50:38.384166Z"},"links":{"cited_paper":"/paper/2305.11747","citing_paper":"/paper/2505.17558"},"observation_digest":"sha256:4c58596e98afcb710ac848a252534bb2ee546ae301b5e7931f823d56a5d42ddd","observation_id":"eef52ccb-362f-41b1-8a75-1afe822787b0","resolution":{"observed_at":"2026-08-07T14:50:38.384166Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.11747","last_updated":"2023-10-23T01:49:32Z","snapshot_observed_at":"2026-08-07T22:56:15.588770Z","submitted_at":"2023-05-19T15:36:27Z","title":"HaluEval: A Large-Scale Hallucination Evaluation Benchmark for Large Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.11747","snapshot_observed_at":"2026-08-07T11:16:25.683786Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.02959","last_updated":"2025-06-03T14:52:44Z","snapshot_observed_at":"2026-08-07T20:33:43.136587Z","submitted_at":"2025-06-03T14:52:44Z","title":"HACo-Det: A Study Towards Fine-Grained Machine-Generated Text Detection under Human-AI Coauthoring","version":1},"reference_index":26,"source":"arxiv_source","source_observed_at":"2026-08-07T11:16:25.683786Z"},"links":{"cited_paper":"/paper/2305.11747","citing_paper":"/paper/2506.02959"},"observation_digest":"sha256:368439ea6e33ab3c78ca18a562f176cd17eb8a8fd9330f7f7b07eb76606751ce","observation_id":"d1caff94-d1e9-41e8-b735-60124e8e6d9c","resolution":{"observed_at":"2026-08-07T11:16:25.683786Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.11747","last_updated":"2023-10-23T01:49:32Z","snapshot_observed_at":"2026-08-07T22:56:15.588770Z","submitted_at":"2023-05-19T15:36:27Z","title":"HaluEval: A Large-Scale Hallucination Evaluation Benchmark for Large Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.11747","snapshot_observed_at":"2026-08-07T05:59:43.915918Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.06539","last_updated":"2025-06-06T21:10:55Z","snapshot_observed_at":"2026-08-07T22:57:24.860316Z","submitted_at":"2025-06-06T21:10:55Z","title":"Beyond Facts: Evaluating Intent Hallucination in Large Language Models","version":1},"reference_index":12,"source":"arxiv_source","source_observed_at":"2026-08-07T05:59:43.915918Z"},"links":{"cited_paper":"/paper/2305.11747","citing_paper":"/paper/2506.06539"},"observation_digest":"sha256:10734fba5e07b13e2d3c6e65f54c6368a122b5c8c528db89ce68506414562658","observation_id":"c0594533-16f9-4a33-a6a3-fb39606e34e9","resolution":{"observed_at":"2026-08-07T05:59:43.915918Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.11747","last_updated":"2023-10-23T01:49:32Z","snapshot_observed_at":"2026-08-07T22:56:15.588770Z","submitted_at":"2023-05-19T15:36:27Z","title":"HaluEval: A Large-Scale Hallucination Evaluation Benchmark for Large Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.11747","snapshot_observed_at":"2026-08-07T10:17:26.745240Z","title":"Halueval: A large-scale hallucination evaluation benchmark for large language models,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.11094","last_updated":"2025-10-30T06:22:33Z","snapshot_observed_at":"2026-08-07T10:11:06.747781Z","submitted_at":"2025-06-06T05:50:50Z","title":"The Scales of Justitia: A Comprehensive Survey on Safety Evaluation of LLMs","version":2},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-07T10:17:26.745240Z"},"links":{"cited_paper":"/paper/2305.11747","citing_paper":"/paper/2506.11094"},"observation_digest":"sha256:0e843dc0d3f1d05c47f5fc30e18dc41a27a067dd336abdddd3b858537170d6f5","observation_id":"ac6ba528-14f1-40f3-8587-189ec7a0cb6d","resolution":{"observed_at":"2026-08-07T10:17:26.745240Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.11747","last_updated":"2023-10-23T01:49:32Z","snapshot_observed_at":"2026-08-07T22:56:15.588770Z","submitted_at":"2023-05-19T15:36:27Z","title":"HaluEval: A Large-Scale Hallucination Evaluation Benchmark for Large Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.11747","snapshot_observed_at":"2026-08-07T00:51:47.094995Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.12538","last_updated":"2025-06-14T15:27:44Z","snapshot_observed_at":"2026-08-07T22:57:24.258176Z","submitted_at":"2025-06-14T15:27:44Z","title":"RealFactBench: A Benchmark for Evaluating Large Language Models in Real-World Fact-Checking","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-07T00:51:47.094995Z"},"links":{"cited_paper":"/paper/2305.11747","citing_paper":"/paper/2506.12538"},"observation_digest":"sha256:e580c7a8ccc28f93db9b86f4201222d0f741d3fa1c9d450f46da59f391d2aa08","observation_id":"917d0631-5793-4d33-90d9-0b2db20f746f","resolution":{"observed_at":"2026-08-07T00:51:47.094995Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.11747","last_updated":"2023-10-23T01:49:32Z","snapshot_observed_at":"2026-08-07T22:56:15.588770Z","submitted_at":"2023-05-19T15:36:27Z","title":"HaluEval: A Large-Scale Hallucination Evaluation Benchmark for Large Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.11747","snapshot_observed_at":"2026-08-07T10:18:46.515677Z","title":"Halueval: A large-scale hallucination evaluation benchmark for large language models,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.02870","last_updated":"2025-06-06T10:50:08Z","snapshot_observed_at":"2026-08-08T01:09:03.836574Z","submitted_at":"2025-06-06T10:50:08Z","title":"Loki's Dance of Illusions: A Comprehensive Survey of Hallucination in Large Language Models","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-07T10:18:46.515677Z"},"links":{"cited_paper":"/paper/2305.11747","citing_paper":"/paper/2507.02870"},"observation_digest":"sha256:445ec16f0880571086dda006c2489af2bae194582811fbb964f2bbcf6d63c782","observation_id":"0616a75f-4787-4f89-97ea-6cddf9a4deda","resolution":{"observed_at":"2026-08-07T10:18:46.515677Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.11747","last_updated":"2023-10-23T01:49:32Z","snapshot_observed_at":"2026-08-07T22:56:15.588770Z","submitted_at":"2023-05-19T15:36:27Z","title":"HaluEval: A Large-Scale Hallucination Evaluation Benchmark for Large Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.11747","snapshot_observed_at":"2026-08-06T14:20:39.469184Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.19586","last_updated":"2025-07-25T18:00:21Z","snapshot_observed_at":"2026-08-07T21:42:42.944892Z","submitted_at":"2025-07-25T18:00:21Z","title":"Mitigating Geospatial Knowledge Hallucination in Large Language Models: Benchmarking and Dynamic Factuality Aligning","version":1},"reference_index":26,"source":"arxiv_source","source_observed_at":"2026-08-06T14:20:39.469184Z"},"links":{"cited_paper":"/paper/2305.11747","citing_paper":"/paper/2507.19586"},"observation_digest":"sha256:d80cd35fbf599533c6c1ecf4688ed03b386e79f53ed61f8f04cf0f5296b956c1","observation_id":"00711774-f145-48f1-b97b-07caf98b913c","resolution":{"observed_at":"2026-08-06T14:20:39.469184Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.11747","last_updated":"2023-10-23T01:49:32Z","snapshot_observed_at":"2026-08-07T22:56:15.588770Z","submitted_at":"2023-05-19T15:36:27Z","title":"HaluEval: A Large-Scale Hallucination Evaluation Benchmark for Large Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.11747","snapshot_observed_at":"2026-08-06T13:56:44.067208Z","title":"doi:10.48550/arXiv.2305.11747 arXiv:2305.11747","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2508.00889","last_updated":"2025-07-26T18:14:18Z","snapshot_observed_at":"2026-08-06T13:56:43.221005Z","submitted_at":"2025-07-26T18:14:18Z","title":"FECT: Factuality Evaluation of Interpretive AI-Generated Claims in Contact Center Conversation Transcripts","version":1},"reference_index":2023,"source":"pdf_text","source_observed_at":"2026-08-06T13:56:44.067208Z"},"links":{"cited_paper":"/paper/2305.11747","citing_paper":"/paper/2508.00889"},"observation_digest":"sha256:2132fdb1cb17e179f3d95115a9c67fc9728f12105e3bad06d274ec1ced296cd7","observation_id":"4d035fcd-6748-45df-b424-add5c791887f","resolution":{"observed_at":"2026-08-06T13:56:44.067208Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.11747","last_updated":"2023-10-23T01:49:32Z","snapshot_observed_at":"2026-08-07T22:56:15.588770Z","submitted_at":"2023-05-19T15:36:27Z","title":"HaluEval: A Large-Scale Hallucination Evaluation Benchmark for Large Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.11747","snapshot_observed_at":"2026-08-05T23:01:47.451336Z","title":"Halueval: A large-scale hallucination evaluation benchmark for large language models.arXiv:2305.11747, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2508.06014","last_updated":"2025-08-08T05:01:17Z","snapshot_observed_at":"2026-08-09T03:19:24.825871Z","submitted_at":"2025-08-08T05:01:17Z","title":"ExploreGS: Explorable 3D Scene Reconstruction with Virtual Camera Samplings and Diffusion Priors","version":1},"reference_index":70,"source":"pdf_text","source_observed_at":"2026-08-05T23:01:47.451336Z"},"links":{"cited_paper":"/paper/2305.11747","citing_paper":"/paper/2508.06014"},"observation_digest":"sha256:d7b88a17bbfaf92d6fbce0d9b44db3cdca0041204c07e35c488de76d585bd4ec","observation_id":"c4296152-11e7-4b14-811f-77ec64f5c1bb","resolution":{"observed_at":"2026-08-05T23:01:47.451336Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.11747","last_updated":"2023-10-23T01:49:32Z","snapshot_observed_at":"2026-08-07T22:56:15.588770Z","submitted_at":"2023-05-19T15:36:27Z","title":"HaluEval: A Large-Scale Hallucination Evaluation Benchmark for Large Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.11747","snapshot_observed_at":"2026-08-05T18:47:45.233361Z","title":"X., Nie, J.-Y","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2508.18235","last_updated":"2025-08-20T00:57:21Z","snapshot_observed_at":"2026-08-05T18:44:01.835548Z","submitted_at":"2025-08-20T00:57:21Z","title":"Sealing The Backdoor: Unlearning Adversarial Text Triggers In Diffusion Models Using Knowledge Distillation","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-05T18:47:45.233361Z"},"links":{"cited_paper":"/paper/2305.11747","citing_paper":"/paper/2508.18235"},"observation_digest":"sha256:a4842f3bd35b85a86c8ab1a76bee52fdf294418253e5ad6bc07b91b8d7b0c41f","observation_id":"6c3719b8-cc2a-418b-83d7-cf561052a6a8","resolution":{"observed_at":"2026-08-05T18:47:45.233361Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.11747","last_updated":"2023-10-23T01:49:32Z","snapshot_observed_at":"2026-08-07T22:56:15.588770Z","submitted_at":"2023-05-19T15:36:27Z","title":"HaluEval: A Large-Scale Hallucination Evaluation Benchmark for Large Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.11747","snapshot_observed_at":"2026-08-05T14:33:18.845937Z","title":"X.; Nie, J.-Y.; and Wen, J.-R","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2508.21228","last_updated":"2025-08-28T21:39:53Z","snapshot_observed_at":"2026-08-06T06:07:09.892510Z","submitted_at":"2025-08-28T21:39:53Z","title":"Decoding Memories: An Efficient Pipeline for Self-Consistency Hallucination Detection","version":1},"reference_index":21,"source":"arxiv_source","source_observed_at":"2026-08-05T14:33:18.845937Z"},"links":{"cited_paper":"/paper/2305.11747","citing_paper":"/paper/2508.21228"},"observation_digest":"sha256:4c9cb9183ea5f6adc3927416f7180706f1e8dd75cab44c4f19fbb405b3f9032c","observation_id":"18ecd53b-ea21-44a8-b9cb-5d1231cafb1b","resolution":{"observed_at":"2026-08-05T14:33:18.845937Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.11747","last_updated":"2023-10-23T01:49:32Z","snapshot_observed_at":"2026-08-07T22:56:15.588770Z","submitted_at":"2023-05-19T15:36:27Z","title":"HaluEval: A Large-Scale Hallucination Evaluation Benchmark for Large Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.11747","snapshot_observed_at":"2026-08-05T10:50:31.830147Z","title":"arXiv:2305.11747 [cs.CL] https://arxiv.org/abs/2305.11747","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2509.05360","last_updated":"2025-09-03T18:52:24Z","snapshot_observed_at":"2026-08-08T02:31:46.854497Z","submitted_at":"2025-09-03T18:52:24Z","title":"Beyond ROUGE: N-Gram Subspace Features for LLM Hallucination Detection","version":1},"reference_index":2023,"source":"pdf_text","source_observed_at":"2026-08-05T10:50:31.830147Z"},"links":{"cited_paper":"/paper/2305.11747","citing_paper":"/paper/2509.05360"},"observation_digest":"sha256:cff5695e078e7e2461882b3d61e0a9ae5a1eae0c39b50d628b62af91324f445c","observation_id":"c4a89746-0f22-4228-98c3-d07851f6bbbe","resolution":{"observed_at":"2026-08-05T10:50:31.830147Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.11747","last_updated":"2023-10-23T01:49:32Z","snapshot_observed_at":"2026-08-07T22:56:15.588770Z","submitted_at":"2023-05-19T15:36:27Z","title":"HaluEval: A Large-Scale Hallucination Evaluation Benchmark for Large Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.11747","snapshot_observed_at":"2026-08-04T20:35:55.014746Z","title":"arXiv:2305.11747 [cs]","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2509.08494","last_updated":"2025-09-10T11:10:10Z","snapshot_observed_at":"2026-08-09T14:03:01.664964Z","submitted_at":"2025-09-10T11:10:10Z","title":"HumanAgencyBench: Scalable Evaluation of Human Agency Support in AI Assistants","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-04T20:35:55.014746Z"},"links":{"cited_paper":"/paper/2305.11747","citing_paper":"/paper/2509.08494"},"observation_digest":"sha256:04cb439b630d0335853a153c67df8535fa806cfe35997c4c7f7669a442aea590","observation_id":"074076d0-24f0-47bd-84f2-ba6105a69771","resolution":{"observed_at":"2026-08-04T20:35:55.014746Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.11747","last_updated":"2023-10-23T01:49:32Z","snapshot_observed_at":"2026-08-07T22:56:15.588770Z","submitted_at":"2023-05-19T15:36:27Z","title":"HaluEval: A Large-Scale Hallucination Evaluation Benchmark for Large Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.11747","snapshot_observed_at":"2026-08-04T22:18:44.151647Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2509.09715","last_updated":"2025-09-09T05:50:08Z","snapshot_observed_at":"2026-08-07T09:17:17.664367Z","submitted_at":"2025-09-09T05:50:08Z","title":"Investigating Symbolic Triggers of Hallucination in Gemma Models Across HaluEval and TruthfulQA","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-04T22:18:44.151647Z"},"links":{"cited_paper":"/paper/2305.11747","citing_paper":"/paper/2509.09715"},"observation_digest":"sha256:e5b2be85e78a45e092e063dab4839db5b4286eaa9bfadb61a4d07de6af0393ce","observation_id":"33d4ea83-1633-4ada-866a-7f811392355a","resolution":{"observed_at":"2026-08-04T22:18:44.151647Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.11747","last_updated":"2023-10-23T01:49:32Z","snapshot_observed_at":"2026-08-07T22:56:15.588770Z","submitted_at":"2023-05-19T15:36:27Z","title":"HaluEval: A Large-Scale Hallucination Evaluation Benchmark for Large Language Models","version":3},"cited_work":{"arxiv_id":"2305.11747","doi":"10.48550/arxiv.2305.11747","metadata_source":"arxiv_reference","pith_arxiv_id":"2305.11747","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv preprint arXiv:2305.11747 (2023)","venue":"arXiv (Cornell University)","work_id":"2cf6bc2d-aed3-4a58-879e-daf0de687940","year":2023},"citing_paper":{"arxiv_id":"2510.07239","last_updated":"2026-05-17T13:01:38Z","snapshot_observed_at":"2026-08-02T19:25:51.433055Z","submitted_at":"2025-10-08T17:06:20Z","title":"Red-Bandit: Test-Time Adaptation for LLM Red-Teaming via Bandit-Guided LoRA Experts","version":2},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-05-21T20:53:58.198974Z"},"links":{"cited_paper":"/paper/2305.11747","citing_paper":"/paper/2510.07239"},"observation_digest":"sha256:eaa16a22a737353c58951bf94f3f11df510e00f95981c7df6053f8d9342f8324","observation_id":"e57d29f1-5bdd-4c49-b3cb-1a233824cbe4","resolution":{"observed_at":"2026-05-21T20:54:21.574843Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.11747","last_updated":"2023-10-23T01:49:32Z","snapshot_observed_at":"2026-08-07T22:56:15.588770Z","submitted_at":"2023-05-19T15:36:27Z","title":"HaluEval: A Large-Scale Hallucination Evaluation Benchmark for Large Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.11747","snapshot_observed_at":"2026-08-03T12:42:35.085906Z","title":"Halueval: A large-scale hallucination evaluation benchmark for large language models,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2601.02023","last_updated":"2026-07-14T21:47:31Z","snapshot_observed_at":"2026-08-08T04:32:46.940287Z","submitted_at":"2026-01-05T11:30:56Z","title":"Not All Needles Are Found: How Fact Distribution and Don't Make It Up Prompts Shape Retrieval, Reasoning, and Hallucination in Long-Context LLMs","version":2},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-08-03T12:42:35.085906Z"},"links":{"cited_paper":"/paper/2305.11747","citing_paper":"/paper/2601.02023"},"observation_digest":"sha256:58c550d6d1d2638bd296c0cbb1b684d310d0a325a192f47c7f5a08411f119a1e","observation_id":"c14198a1-4884-413e-8e8d-e25c7366b5f9","resolution":{"observed_at":"2026-08-03T12:42:35.085906Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.11747","last_updated":"2023-10-23T01:49:32Z","snapshot_observed_at":"2026-08-07T22:56:15.588770Z","submitted_at":"2023-05-19T15:36:27Z","title":"HaluEval: A Large-Scale Hallucination Evaluation Benchmark for Large Language Models","version":3},"cited_work":{"arxiv_id":"2305.11747","doi":"10.48550/arxiv.2305.11747","metadata_source":"arxiv_reference","pith_arxiv_id":"2305.11747","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv preprint arXiv:2305.11747 (2023)","venue":"arXiv (Cornell University)","work_id":"2cf6bc2d-aed3-4a58-879e-daf0de687940","year":2023},"citing_paper":{"arxiv_id":"2601.03846","last_updated":"2026-04-19T20:12:38Z","snapshot_observed_at":"2026-07-06T22:41:01.254450Z","submitted_at":"2026-01-07T12:07:48Z","title":"When Numbers Start Talking: Implicit Numerical Coordination Among LLM-Based Agents","version":2},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-05-16T16:39:34.129361Z"},"links":{"cited_paper":"/paper/2305.11747","citing_paper":"/paper/2601.03846"},"observation_digest":"sha256:e0e68bed5d45d7b7064140af5694b334a560a73efe62474c522ff351b49f5223","observation_id":"131a42ec-7df5-4cf0-8d20-7d23abb9f237","resolution":{"observed_at":"2026-05-16T16:41:06.170142Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.11747","last_updated":"2023-10-23T01:49:32Z","snapshot_observed_at":"2026-08-07T22:56:15.588770Z","submitted_at":"2023-05-19T15:36:27Z","title":"HaluEval: A Large-Scale Hallucination Evaluation Benchmark for Large Language Models","version":3},"cited_work":{"arxiv_id":"2305.11747","doi":"10.48550/arxiv.2305.11747","metadata_source":"arxiv_reference","pith_arxiv_id":"2305.11747","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv preprint arXiv:2305.11747 (2023)","venue":"arXiv (Cornell University)","work_id":"2cf6bc2d-aed3-4a58-879e-daf0de687940","year":2023},"citing_paper":{"arxiv_id":"2602.06718","last_updated":"2026-05-14T09:48:40Z","snapshot_observed_at":"2026-07-06T22:44:51.126114Z","submitted_at":"2026-02-06T14:08:34Z","title":"GhostCite: A Large-Scale Analysis of Citation Validity in the Age of Large Language Models","version":2},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-05-16T07:00:41.337663Z"},"links":{"cited_paper":"/paper/2305.11747","citing_paper":"/paper/2602.06718"},"observation_digest":"sha256:993b290328e515689453fc4ed7a8ab0384d1e0368c5b3ce4ac134001d6d105b8","observation_id":"9707cb7f-61cd-4609-97eb-f4797891b0df","resolution":{"observed_at":"2026-05-16T07:00:42.679785Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.11747","last_updated":"2023-10-23T01:49:32Z","snapshot_observed_at":"2026-08-07T22:56:15.588770Z","submitted_at":"2023-05-19T15:36:27Z","title":"HaluEval: A Large-Scale Hallucination Evaluation Benchmark for Large Language Models","version":3},"cited_work":{"arxiv_id":"2305.11747","doi":"10.48550/arxiv.2305.11747","metadata_source":"arxiv_reference","pith_arxiv_id":"2305.11747","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv preprint arXiv:2305.11747 (2023)","venue":"arXiv (Cornell University)","work_id":"2cf6bc2d-aed3-4a58-879e-daf0de687940","year":2023},"citing_paper":{"arxiv_id":"2603.14987","last_updated":"2026-05-21T06:24:06Z","snapshot_observed_at":"2026-08-03T21:30:30.993382Z","submitted_at":"2026-03-16T08:51:33Z","title":"Beyond Benchmark Islands: Toward Representative Trustworthiness Evaluation for Agentic AI","version":2},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-05-22T10:19:56.003219Z"},"links":{"cited_paper":"/paper/2305.11747","citing_paper":"/paper/2603.14987"},"observation_digest":"sha256:720df485703d45780fb33d295cfec57baac09ea3311ed952c3110018dce3ac83","observation_id":"596840b4-65ce-4ef8-ae82-b1ae0a9a0564","resolution":{"observed_at":"2026-05-22T10:21:23.218723Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.11747","last_updated":"2023-10-23T01:49:32Z","snapshot_observed_at":"2026-08-07T22:56:15.588770Z","submitted_at":"2023-05-19T15:36:27Z","title":"HaluEval: A Large-Scale Hallucination Evaluation Benchmark for Large Language Models","version":3},"cited_work":{"arxiv_id":"2305.11747","doi":"10.48550/arxiv.2305.11747","metadata_source":"arxiv_reference","pith_arxiv_id":"2305.11747","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv preprint arXiv:2305.11747 (2023)","venue":"arXiv (Cornell University)","work_id":"2cf6bc2d-aed3-4a58-879e-daf0de687940","year":2023},"citing_paper":{"arxiv_id":"2604.04743","last_updated":"2026-04-06T15:08:54Z","snapshot_observed_at":"2026-08-02T07:37:15.205748Z","submitted_at":"2026-04-06T15:08:54Z","title":"Hallucination Basins: A Dynamic Framework for Understanding and Controlling LLM Hallucinations","version":1},"reference_index":13,"source":"arxiv_source","source_observed_at":"2026-05-10T18:57:00.087829Z"},"links":{"cited_paper":"/paper/2305.11747","citing_paper":"/paper/2604.04743"},"observation_digest":"sha256:9978e8865f719a412f07465f8455508dfde67e6ed3bf03ecfc731a1bcfbc6424","observation_id":"0057c348-a62d-46cb-8728-958a9dbd383a","resolution":{"observed_at":"2026-05-10T23:40:52.241366Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.11747","last_updated":"2023-10-23T01:49:32Z","snapshot_observed_at":"2026-08-07T22:56:15.588770Z","submitted_at":"2023-05-19T15:36:27Z","title":"HaluEval: A Large-Scale Hallucination Evaluation Benchmark for Large Language Models","version":3},"cited_work":{"arxiv_id":"2305.11747","doi":"10.48550/arxiv.2305.11747","metadata_source":"arxiv_reference","pith_arxiv_id":"2305.11747","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv preprint arXiv:2305.11747 (2023)","venue":"arXiv (Cornell University)","work_id":"2cf6bc2d-aed3-4a58-879e-daf0de687940","year":2023},"citing_paper":{"arxiv_id":"2604.08519","last_updated":"2026-04-09T17:55:50Z","snapshot_observed_at":"2026-07-06T22:57:35.435713Z","submitted_at":"2026-04-09T17:55:50Z","title":"Cram Less to Fit More: Training Data Pruning Improves Memorization of Facts","version":1},"reference_index":50,"source":"arxiv_source","source_observed_at":"2026-05-10T17:42:31.465077Z"},"links":{"cited_paper":"/paper/2305.11747","citing_paper":"/paper/2604.08519"},"observation_digest":"sha256:54d6d1f055c8793b9d9b9280322b6e8e96a086c6e7db00ad6497b01ca87824e8","observation_id":"2b179962-ea43-420d-85d1-d7be6a086cd1","resolution":{"observed_at":"2026-05-11T06:15:59.084616Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.11747","last_updated":"2023-10-23T01:49:32Z","snapshot_observed_at":"2026-08-07T22:56:15.588770Z","submitted_at":"2023-05-19T15:36:27Z","title":"HaluEval: A Large-Scale Hallucination Evaluation Benchmark for Large Language Models","version":3},"cited_work":{"arxiv_id":"2305.11747","doi":"10.48550/arxiv.2305.11747","metadata_source":"arxiv_reference","pith_arxiv_id":"2305.11747","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv preprint arXiv:2305.11747 (2023)","venue":"arXiv (Cornell University)","work_id":"2cf6bc2d-aed3-4a58-879e-daf0de687940","year":2023},"citing_paper":{"arxiv_id":"2604.10733","last_updated":"2026-04-12T17:12:55Z","snapshot_observed_at":"2026-07-06T22:59:18.756089Z","submitted_at":"2026-04-12T17:12:55Z","title":"Too Nice to Tell the Truth: Quantifying Agreeableness-Driven Sycophancy in Role-Playing Language Models","version":1},"reference_index":29,"source":"arxiv_source","source_observed_at":"2026-05-10T16:05:09.033412Z"},"links":{"cited_paper":"/paper/2305.11747","citing_paper":"/paper/2604.10733"},"observation_digest":"sha256:b86d41a491116bad2ea9a0f2b5785d44440f2a0d11fcfce2038da4ad6acde004","observation_id":"09b24ac1-4f0f-44e7-b8a7-061460ee66e8","resolution":{"observed_at":"2026-05-11T09:21:00.939630Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.11747","last_updated":"2023-10-23T01:49:32Z","snapshot_observed_at":"2026-08-07T22:56:15.588770Z","submitted_at":"2023-05-19T15:36:27Z","title":"HaluEval: A Large-Scale Hallucination Evaluation Benchmark for Large Language Models","version":3},"cited_work":{"arxiv_id":"2305.11747","doi":"10.48550/arxiv.2305.11747","metadata_source":"arxiv_reference","pith_arxiv_id":"2305.11747","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv preprint arXiv:2305.11747 (2023)","venue":"arXiv (Cornell University)","work_id":"2cf6bc2d-aed3-4a58-879e-daf0de687940","year":2023},"citing_paper":{"arxiv_id":"2604.10734","last_updated":"2026-04-12T17:14:36Z","snapshot_observed_at":"2026-07-06T22:59:18.756089Z","submitted_at":"2026-04-12T17:14:36Z","title":"Self-Correcting RAG: Enhancing Faithfulness via MMKP Context Selection and NLI-Guided MCTS","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-10T15:58:15.150613Z"},"links":{"cited_paper":"/paper/2305.11747","citing_paper":"/paper/2604.10734"},"observation_digest":"sha256:b1976ea9eaee39135fcfe16710ea13566b527d9d3febe7d675e21d5450b22cf3","observation_id":"4bee8821-3fb9-423f-acb1-331238bfe827","resolution":{"observed_at":"2026-05-11T09:31:05.190349Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.11747","last_updated":"2023-10-23T01:49:32Z","snapshot_observed_at":"2026-08-07T22:56:15.588770Z","submitted_at":"2023-05-19T15:36:27Z","title":"HaluEval: A Large-Scale Hallucination Evaluation Benchmark for Large Language Models","version":3},"cited_work":{"arxiv_id":"2305.11747","doi":"10.48550/arxiv.2305.11747","metadata_source":"arxiv_reference","pith_arxiv_id":"2305.11747","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv preprint arXiv:2305.11747 (2023)","venue":"arXiv (Cornell University)","work_id":"2cf6bc2d-aed3-4a58-879e-daf0de687940","year":2023},"citing_paper":{"arxiv_id":"2604.15945","last_updated":"2026-04-17T11:07:32Z","snapshot_observed_at":"2026-07-06T23:03:21.422544Z","submitted_at":"2026-04-17T11:07:32Z","title":"RAGognizer: Hallucination-Aware Fine-Tuning via Detection Head Integration","version":1},"reference_index":28,"source":"arxiv_source","source_observed_at":"2026-05-10T08:38:42.029762Z"},"links":{"cited_paper":"/paper/2305.11747","citing_paper":"/paper/2604.15945"},"observation_digest":"sha256:0b6d91a3334bd3f4e0550f3f53a5f026793e0e2bf4d4cf521f0fe5dc4599fc8e","observation_id":"cbb8b087-7bba-4429-99cb-bdd18d0757e3","resolution":{"observed_at":"2026-05-10T08:43:02.040437Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.11747","last_updated":"2023-10-23T01:49:32Z","snapshot_observed_at":"2026-08-07T22:56:15.588770Z","submitted_at":"2023-05-19T15:36:27Z","title":"HaluEval: A Large-Scale Hallucination Evaluation Benchmark for Large Language Models","version":3},"cited_work":{"arxiv_id":"2305.11747","doi":"10.48550/arxiv.2305.11747","metadata_source":"arxiv_reference","pith_arxiv_id":"2305.11747","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv preprint arXiv:2305.11747 (2023)","venue":"arXiv (Cornell University)","work_id":"2cf6bc2d-aed3-4a58-879e-daf0de687940","year":2023},"citing_paper":{"arxiv_id":"2605.02443","last_updated":"2026-05-22T15:24:30Z","snapshot_observed_at":"2026-08-02T13:58:05.076867Z","submitted_at":"2026-05-04T10:43:27Z","title":"HalluScan: A Systematic Benchmark for Detecting and Mitigating Hallucinations in Instruction-Following LLMs","version":2},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-05-25T06:49:16.755597Z"},"links":{"cited_paper":"/paper/2305.11747","citing_paper":"/paper/2605.02443"},"observation_digest":"sha256:5550a358cc39018ae145a79e94fec987dd6395538fca36d5e5126749a64ac27b","observation_id":"657a417e-3833-43d4-8854-a9151f0d8e85","resolution":{"observed_at":"2026-05-25T06:50:28.037822Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.11747","last_updated":"2023-10-23T01:49:32Z","snapshot_observed_at":"2026-08-07T22:56:15.588770Z","submitted_at":"2023-05-19T15:36:27Z","title":"HaluEval: A Large-Scale Hallucination Evaluation Benchmark for Large Language Models","version":3},"cited_work":{"arxiv_id":"2305.11747","doi":"10.48550/arxiv.2305.11747","metadata_source":"arxiv_reference","pith_arxiv_id":"2305.11747","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv preprint arXiv:2305.11747 (2023)","venue":"arXiv (Cornell University)","work_id":"2cf6bc2d-aed3-4a58-879e-daf0de687940","year":2023},"citing_paper":{"arxiv_id":"2605.05810","last_updated":"2026-05-07T07:46:17Z","snapshot_observed_at":"2026-07-06T23:18:22.301347Z","submitted_at":"2026-05-07T07:46:17Z","title":"CXR-ContraBench: Benchmarking Negated-Option Attraction in Medical VLMs","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-05-08T14:49:53.357083Z"},"links":{"cited_paper":"/paper/2305.11747","citing_paper":"/paper/2605.05810"},"observation_digest":"sha256:9c01b83753075fc05120239d2ea42bfee990c9115f06b3b643074d5beaac407d","observation_id":"bc60a18a-ddd8-497b-a0b9-d156d92e05d5","resolution":{"observed_at":"2026-05-11T18:41:09.490126Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.11747","last_updated":"2023-10-23T01:49:32Z","snapshot_observed_at":"2026-08-07T22:56:15.588770Z","submitted_at":"2023-05-19T15:36:27Z","title":"HaluEval: A Large-Scale Hallucination Evaluation Benchmark for Large Language Models","version":3},"cited_work":{"arxiv_id":"2305.11747","doi":"10.48550/arxiv.2305.11747","metadata_source":"arxiv_reference","pith_arxiv_id":"2305.11747","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv preprint arXiv:2305.11747 (2023)","venue":"arXiv (Cornell University)","work_id":"2cf6bc2d-aed3-4a58-879e-daf0de687940","year":2023},"citing_paper":{"arxiv_id":"2605.17007","last_updated":"2026-05-16T14:08:15Z","snapshot_observed_at":"2026-07-06T23:28:02.429673Z","submitted_at":"2026-05-16T14:08:15Z","title":"HalluScore: Large Language Model Hallucination Question Answering Benchmark","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-05-19T20:31:20.017866Z"},"links":{"cited_paper":"/paper/2305.11747","citing_paper":"/paper/2605.17007"},"observation_digest":"sha256:e1d8778b5776aaf87759de3bd817ef79c1ff3b974bf100b3760cc7919c63192d","observation_id":"e244d198-3317-4186-b063-28b3a535b664","resolution":{"observed_at":"2026-05-19T20:32:45.413600Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.11747","last_updated":"2023-10-23T01:49:32Z","snapshot_observed_at":"2026-08-07T22:56:15.588770Z","submitted_at":"2023-05-19T15:36:27Z","title":"HaluEval: A Large-Scale Hallucination Evaluation Benchmark for Large Language Models","version":3},"cited_work":{"arxiv_id":"2305.11747","doi":"10.48550/arxiv.2305.11747","metadata_source":"arxiv_reference","pith_arxiv_id":"2305.11747","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv preprint arXiv:2305.11747 (2023)","venue":"arXiv (Cornell University)","work_id":"2cf6bc2d-aed3-4a58-879e-daf0de687940","year":2023},"citing_paper":{"arxiv_id":"2605.23262","last_updated":"2026-05-22T06:03:01Z","snapshot_observed_at":"2026-08-02T00:22:40.506306Z","submitted_at":"2026-05-22T06:03:01Z","title":"Design and Report Benchmarks for Knowledge Work","version":1},"reference_index":61,"source":"arxiv_source","source_observed_at":"2026-05-25T04:39:14.319133Z"},"links":{"cited_paper":"/paper/2305.11747","citing_paper":"/paper/2605.23262"},"observation_digest":"sha256:8788886849aba927d918251f34873af04f033ab6f101417b7a65b7fae941bbb4","observation_id":"4c646df3-dd00-4251-aaf9-ff4d5a7d5a5c","resolution":{"observed_at":"2026-05-25T04:40:23.520503Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.11747","last_updated":"2023-10-23T01:49:32Z","snapshot_observed_at":"2026-08-07T22:56:15.588770Z","submitted_at":"2023-05-19T15:36:27Z","title":"HaluEval: A Large-Scale Hallucination Evaluation Benchmark for Large Language Models","version":3},"cited_work":{"arxiv_id":"2305.11747","doi":"10.48550/arxiv.2305.11747","metadata_source":"arxiv_reference","pith_arxiv_id":"2305.11747","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv preprint arXiv:2305.11747 (2023)","venue":"arXiv (Cornell University)","work_id":"2cf6bc2d-aed3-4a58-879e-daf0de687940","year":2023},"citing_paper":{"arxiv_id":"2605.24919","last_updated":"2026-05-24T07:50:03Z","snapshot_observed_at":"2026-08-09T11:57:25.465812Z","submitted_at":"2026-05-24T07:50:03Z","title":"MultiHaluDet: Multilingual Hallucination Detection via LLM Hidden State Probing","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-06-30T12:29:59.165791Z"},"links":{"cited_paper":"/paper/2305.11747","citing_paper":"/paper/2605.24919"},"observation_digest":"sha256:b630b03eafb1b594f209cfb63b5469c9c6df369922e87e15d3e7cba15f1c8844","observation_id":"4f0a9440-709a-4989-89ec-5a48a97d5838","resolution":{"observed_at":"2026-06-30T12:34:38.644555Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.11747","last_updated":"2023-10-23T01:49:32Z","snapshot_observed_at":"2026-08-07T22:56:15.588770Z","submitted_at":"2023-05-19T15:36:27Z","title":"HaluEval: A Large-Scale Hallucination Evaluation Benchmark for Large Language Models","version":3},"cited_work":{"arxiv_id":"2305.11747","doi":"10.48550/arxiv.2305.11747","metadata_source":"arxiv_reference","pith_arxiv_id":"2305.11747","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv preprint arXiv:2305.11747 (2023)","venue":"arXiv (Cornell University)","work_id":"2cf6bc2d-aed3-4a58-879e-daf0de687940","year":2023},"citing_paper":{"arxiv_id":"2605.28778","last_updated":"2026-05-27T17:38:00Z","snapshot_observed_at":"2026-07-06T23:38:19.096349Z","submitted_at":"2026-05-27T17:38:00Z","title":"Can LLMs Use Linguistic Uncertainty Markers to Reliably Reflect Intrinsic Confidence?","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-06-29T12:18:36.854164Z"},"links":{"cited_paper":"/paper/2305.11747","citing_paper":"/paper/2605.28778"},"observation_digest":"sha256:e227a9833a16fc29b235dbe6888fe578a1f3dff71f388f690fcd345ebae66a9a","observation_id":"3c55d3a7-6110-48c8-8451-e0ba570521a9","resolution":{"observed_at":"2026-06-29T12:23:24.355085Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.11747","last_updated":"2023-10-23T01:49:32Z","snapshot_observed_at":"2026-08-07T22:56:15.588770Z","submitted_at":"2023-05-19T15:36:27Z","title":"HaluEval: A Large-Scale Hallucination Evaluation Benchmark for Large Language Models","version":3},"cited_work":{"arxiv_id":"2305.11747","doi":"10.48550/arxiv.2305.11747","metadata_source":"arxiv_reference","pith_arxiv_id":"2305.11747","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv preprint arXiv:2305.11747 (2023)","venue":"arXiv (Cornell University)","work_id":"2cf6bc2d-aed3-4a58-879e-daf0de687940","year":2023},"citing_paper":{"arxiv_id":"2605.29523","last_updated":"2026-05-28T07:40:19Z","snapshot_observed_at":"2026-08-06T18:48:51.326147Z","submitted_at":"2026-05-28T07:40:19Z","title":"K-FinHallu: A Hallucination Detection Benchmark for Multi-Turn RAG in Korean Finance","version":1},"reference_index":17,"source":"arxiv_source","source_observed_at":"2026-06-29T08:54:48.807164Z"},"links":{"cited_paper":"/paper/2305.11747","citing_paper":"/paper/2605.29523"},"observation_digest":"sha256:58e49f248b00c04837d6e811ff5bda328f62c9d8027280172978d47b8621ef54","observation_id":"42d5f45e-2169-47e8-b63d-3bc7f987349b","resolution":{"observed_at":"2026-06-29T09:03:16.135655Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.11747","last_updated":"2023-10-23T01:49:32Z","snapshot_observed_at":"2026-08-07T22:56:15.588770Z","submitted_at":"2023-05-19T15:36:27Z","title":"HaluEval: A Large-Scale Hallucination Evaluation Benchmark for Large Language Models","version":3},"cited_work":{"arxiv_id":"2305.11747","doi":"10.48550/arxiv.2305.11747","metadata_source":"arxiv_reference","pith_arxiv_id":"2305.11747","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv preprint arXiv:2305.11747 (2023)","venue":"arXiv (Cornell University)","work_id":"2cf6bc2d-aed3-4a58-879e-daf0de687940","year":2023},"citing_paper":{"arxiv_id":"2606.04435","last_updated":"2026-06-03T04:33:47Z","snapshot_observed_at":"2026-07-06T23:44:33.095008Z","submitted_at":"2026-06-03T04:33:47Z","title":"Cascading Hallucination in Agentic RAG: The CHARM Framework for Detection and Mitigation","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-06-28T06:33:11.701246Z"},"links":{"cited_paper":"/paper/2305.11747","citing_paper":"/paper/2606.04435"},"observation_digest":"sha256:eaee1f752af72a950f9967d2efeb0e3119de9e4e0311863a2a9f651d4cc037cd","observation_id":"4219f27f-bf66-4e6e-a05d-9a30eb551044","resolution":{"observed_at":"2026-07-02T07:56:47.451220Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.11747","last_updated":"2023-10-23T01:49:32Z","snapshot_observed_at":"2026-08-07T22:56:15.588770Z","submitted_at":"2023-05-19T15:36:27Z","title":"HaluEval: A Large-Scale Hallucination Evaluation Benchmark for Large Language Models","version":3},"cited_work":{"arxiv_id":"2305.11747","doi":"10.48550/arxiv.2305.11747","metadata_source":"arxiv_reference","pith_arxiv_id":"2305.11747","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv preprint arXiv:2305.11747 (2023)","venue":"arXiv (Cornell University)","work_id":"2cf6bc2d-aed3-4a58-879e-daf0de687940","year":2023},"citing_paper":{"arxiv_id":"2606.08705","last_updated":"2026-06-07T16:03:14Z","snapshot_observed_at":"2026-08-06T10:14:38.045879Z","submitted_at":"2026-06-07T16:03:14Z","title":"Analyzing the Correlation Between Hallucinations and Knowledge Conflicts in Large Language Models","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-06-27T18:43:58.988270Z"},"links":{"cited_paper":"/paper/2305.11747","citing_paper":"/paper/2606.08705"},"observation_digest":"sha256:39d99a72ffcc9d78212edaa2f1ebca9675c3abe263069a9226d0aee4afc1774d","observation_id":"42723d12-efeb-47cf-8361-a050b16ffdc1","resolution":{"observed_at":"2026-07-02T22:37:26.443815Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.11747","last_updated":"2023-10-23T01:49:32Z","snapshot_observed_at":"2026-08-07T22:56:15.588770Z","submitted_at":"2023-05-19T15:36:27Z","title":"HaluEval: A Large-Scale Hallucination Evaluation Benchmark for Large Language Models","version":3},"cited_work":{"arxiv_id":"2305.11747","doi":"10.48550/arxiv.2305.11747","metadata_source":"arxiv_reference","pith_arxiv_id":"2305.11747","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv preprint arXiv:2305.11747 (2023)","venue":"arXiv (Cornell University)","work_id":"2cf6bc2d-aed3-4a58-879e-daf0de687940","year":2023},"citing_paper":{"arxiv_id":"2606.13220","last_updated":"2026-06-11T11:37:07Z","snapshot_observed_at":"2026-07-06T23:51:59.619032Z","submitted_at":"2026-06-11T11:37:07Z","title":"LLM-as-an-Investigator: Evidence-First Reasoning for Robust Interactive Problem Diagnosis","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-06-27T06:58:42.823851Z"},"links":{"cited_paper":"/paper/2305.11747","citing_paper":"/paper/2606.13220"},"observation_digest":"sha256:e3b9856e09c42e575c11b3ebc79f4eb6b26e700d5d460b151299a75f7ba09c9a","observation_id":"fafa27c4-dae0-4fea-8701-446acae67341","resolution":{"observed_at":"2026-07-03T14:38:29.256488Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.11747","last_updated":"2023-10-23T01:49:32Z","snapshot_observed_at":"2026-08-07T22:56:15.588770Z","submitted_at":"2023-05-19T15:36:27Z","title":"HaluEval: A Large-Scale Hallucination Evaluation Benchmark for Large Language Models","version":3},"cited_work":{"arxiv_id":"2305.11747","doi":"10.48550/arxiv.2305.11747","metadata_source":"arxiv_reference","pith_arxiv_id":"2305.11747","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv preprint arXiv:2305.11747 (2023)","venue":"arXiv (Cornell University)","work_id":"2cf6bc2d-aed3-4a58-879e-daf0de687940","year":2023},"citing_paper":{"arxiv_id":"2606.32032","last_updated":"2026-06-30T17:56:01Z","snapshot_observed_at":"2026-07-07T00:05:41.920778Z","submitted_at":"2026-06-30T17:56:01Z","title":"Reinforcement Learning with Metacognitive Feedback Elicits Faithful Uncertainty Expression in LLMs","version":1},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-07-01T05:22:38.232552Z"},"links":{"cited_paper":"/paper/2305.11747","citing_paper":"/paper/2606.32032"},"observation_digest":"sha256:2651d73f816ce70ce406beeb2236b5bad54a7612340baa79529bd24cde85d721","observation_id":"00832ebc-def9-41d3-96d5-1bcd38389bca","resolution":{"observed_at":"2026-07-01T10:35:42.138233Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-09T06:31:02.800959+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.11747","last_updated":"2023-10-23T01:49:32Z","snapshot_observed_at":"2026-08-07T22:56:15.588770Z","submitted_at":"2023-05-19T15:36:27Z","title":"HaluEval: A Large-Scale Hallucination Evaluation Benchmark for Large Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.11747","snapshot_observed_at":"2026-07-14T05:45:08.896651Z","title":"InProceedings of the 2023 Conference on Empirical Methods in Natural Language Processing (EMNLP)","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.11414","last_updated":"2026-07-13T11:22:25Z","snapshot_observed_at":"2026-08-07T10:35:43.300492Z","submitted_at":"2026-07-13T11:22:25Z","title":"Confidently Wrong: Detecting Hallucinations in Financial Question Answering from LLM Internal States","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-07-14T05:45:08.896651Z"},"links":{"cited_paper":"/paper/2305.11747","citing_paper":"/paper/2607.11414"},"observation_digest":"sha256:41459fdac211997f155e455cf3f47d647e51e37d93dfc20bbde3ad193c850673","observation_id":"83edaeed-4c33-4fa0-be17-a7fd669a0976","resolution":{"observed_at":"2026-07-14T05:45:08.896651Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.11747","last_updated":"2023-10-23T01:49:32Z","snapshot_observed_at":"2026-08-07T22:56:15.588770Z","submitted_at":"2023-05-19T15:36:27Z","title":"HaluEval: A Large-Scale Hallucination Evaluation Benchmark for Large Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.11747","snapshot_observed_at":"2026-08-02T04:30:14.920973Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.13707","last_updated":"2026-07-15T11:12:54Z","snapshot_observed_at":"2026-08-07T22:58:24.972946Z","submitted_at":"2026-07-15T11:12:54Z","title":"The Test Oracle Problem in Synthetic LLM-as-Judge Corpora: Disappearance, Distortion and a Validation Protocol","version":1},"reference_index":2023,"source":"pdf_text","source_observed_at":"2026-08-02T04:30:14.920973Z"},"links":{"cited_paper":"/paper/2305.11747","citing_paper":"/paper/2607.13707"},"observation_digest":"sha256:f1b35fde073a946a32fbc891991a4d6510bf945cbfd3fc5ba62ffd5316070cf8","observation_id":"da4e7f99-9b59-4835-b4b3-4818e0f03a67","resolution":{"observed_at":"2026-08-02T04:30:14.920973Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.11747","last_updated":"2023-10-23T01:49:32Z","snapshot_observed_at":"2026-08-07T22:56:15.588770Z","submitted_at":"2023-05-19T15:36:27Z","title":"HaluEval: A Large-Scale Hallucination Evaluation Benchmark for Large Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.11747","snapshot_observed_at":"2026-08-02T03:41:02.990486Z","title":"arXiv preprint arXiv:2305.11747 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.13820","last_updated":"2026-07-15T13:28:18Z","snapshot_observed_at":"2026-08-08T03:14:24.536574Z","submitted_at":"2026-07-15T13:28:18Z","title":"PROBE: Benchmarking Code Generation in Large Language Models","version":1},"reference_index":59,"source":"arxiv_source","source_observed_at":"2026-08-02T03:41:02.990486Z"},"links":{"cited_paper":"/paper/2305.11747","citing_paper":"/paper/2607.13820"},"observation_digest":"sha256:7a29a2dfa09b580302fef2c2548c4d8021a79f2658c589160e018746e83962e2","observation_id":"fc2a8aa5-5343-4316-9cc9-3159358ed8aa","resolution":{"observed_at":"2026-08-02T03:41:02.990486Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.11747","last_updated":"2023-10-23T01:49:32Z","snapshot_observed_at":"2026-08-07T22:56:15.588770Z","submitted_at":"2023-05-19T15:36:27Z","title":"HaluEval: A Large-Scale Hallucination Evaluation Benchmark for Large Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.11747","snapshot_observed_at":"2026-08-02T09:19:44.989763Z","title":"doi:10.48550/arXiv.2305.11747 , url =","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.18292","last_updated":"2026-07-28T17:58:05Z","snapshot_observed_at":"2026-08-08T06:05:54.064255Z","submitted_at":"2026-06-30T17:51:04Z","title":"Reliability Scales Inversely: Hallucinations Snowball Faster in Bigger Language Models","version":3},"reference_index":47,"source":"arxiv_source","source_observed_at":"2026-08-02T09:19:44.989763Z"},"links":{"cited_paper":"/paper/2305.11747","citing_paper":"/paper/2607.18292"},"observation_digest":"sha256:24a11c882018ca77d466f5e3c06b7a0d2b92d83113b631ad544832765ed5a2fb","observation_id":"035824ea-f668-4ebb-a588-122af89fabc2","resolution":{"observed_at":"2026-08-02T09:19:44.989763Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.11747","last_updated":"2023-10-23T01:49:32Z","snapshot_observed_at":"2026-08-07T22:56:15.588770Z","submitted_at":"2023-05-19T15:36:27Z","title":"HaluEval: A Large-Scale Hallucination Evaluation Benchmark for Large Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.11747","snapshot_observed_at":"2026-08-01T14:01:26.820493Z","title":"inclusionAI","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.18915","last_updated":"2026-07-21T09:56:58Z","snapshot_observed_at":"2026-08-07T05:56:25.485439Z","submitted_at":"2026-07-21T09:56:58Z","title":"Reasoning Error from Known Fact: Step-Level Self-Consistency Group Relative Policy Optimization for LLM","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-01T14:01:26.820493Z"},"links":{"cited_paper":"/paper/2305.11747","citing_paper":"/paper/2607.18915"},"observation_digest":"sha256:e4573b817dfa9c6f427459f5b1a86dfe2482fca6c17b50ef7f29b53d64dad58b","observation_id":"76f21532-3274-4787-8d51-d5a7ca7eb161","resolution":{"observed_at":"2026-08-01T14:01:26.820493Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.11747","last_updated":"2023-10-23T01:49:32Z","snapshot_observed_at":"2026-08-07T22:56:15.588770Z","submitted_at":"2023-05-19T15:36:27Z","title":"HaluEval: A Large-Scale Hallucination Evaluation Benchmark for Large Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.11747","snapshot_observed_at":"2026-08-01T07:12:17.577089Z","title":"arXiv preprint arXiv:2305.11747 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.21550","last_updated":"2026-07-23T17:35:20Z","snapshot_observed_at":"2026-08-08T08:48:36.880078Z","submitted_at":"2026-07-23T17:35:20Z","title":"X$^3$-OPD: Distilling Reasoning into Large Audio-Language Models via On-Policy Alignment","version":1},"reference_index":106,"source":"arxiv_source","source_observed_at":"2026-08-01T07:12:17.577089Z"},"links":{"cited_paper":"/paper/2305.11747","citing_paper":"/paper/2607.21550"},"observation_digest":"sha256:44a46ff62c7079ae1be13144c83caa579942177c74667372284158595276e642","observation_id":"ccb68686-ef83-4ab6-90c1-e65570b663bd","resolution":{"observed_at":"2026-08-01T07:12:17.577089Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.11747","last_updated":"2023-10-23T01:49:32Z","snapshot_observed_at":"2026-08-07T22:56:15.588770Z","submitted_at":"2023-05-19T15:36:27Z","title":"HaluEval: A Large-Scale Hallucination Evaluation Benchmark for Large Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.11747","snapshot_observed_at":"2026-08-07T22:53:05.080981Z","title":"arXiv preprint arXiv:2305.11747 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.05823","last_updated":"2026-08-06T09:52:09Z","snapshot_observed_at":"2026-08-09T14:11:03.315519Z","submitted_at":"2026-08-06T09:52:09Z","title":"Decomposed Entailment for Factuality Checking and Hallucination Detection","version":1},"reference_index":14,"source":"arxiv_source","source_observed_at":"2026-08-07T22:53:05.080981Z"},"links":{"cited_paper":"/paper/2305.11747","citing_paper":"/paper/2608.05823"},"observation_digest":"sha256:da0e6a4c129f09bce642e9cdede24036fed9fa59b063021bcb32223b52f8407b","observation_id":"5612b22d-9735-422c-954e-09e0dd4a580c","resolution":{"observed_at":"2026-08-07T22:53:05.080981Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2305.11747/citation-record","integrity":"/paper/2305.11747/integrity","json":"/paper/2305.11747/citation-record.json","paper":"/paper/2305.11747"},"outbound":[],"paper":{"arxiv_id":"2305.11747","last_updated":"2023-10-23T01:49:32Z","latest_version":3,"primary_category":"cs.CL","snapshot_observed_at":"2026-08-07T22:56:15.588770Z","submitted_at":"2023-05-19T15:36:27Z","title":"HaluEval: A Large-Scale Hallucination Evaluation Benchmark for Large Language Models"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"thesis":"As of 9 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 52 inbound Pith citation observations for arXiv:2305.11747."}