{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:XJTUURHHYDAUATWT4SAFQQRWJH","short_pith_number":"pith:XJTUURHH","schema_version":"1.0","canonical_sha256":"ba674a44e7c0c1404ed3e48058423649e60088818993ebab9995a1576da94857","source":{"kind":"arxiv","id":"2407.08488","version":2},"attestation_state":"computed","paper":{"title":"Lynx: An Open Source Hallucination Evaluation Model","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.AI","authors_text":"Anand Kannappan, Bartosz Mielczarek, Douwe Kiela, Rebecca Qian, Selvan Sunitha Ravi","submitted_at":"2024-07-11T13:22:17Z","abstract_excerpt":"Retrieval Augmented Generation (RAG) techniques aim to mitigate hallucinations in Large Language Models (LLMs). However, LLMs can still produce information that is unsupported or contradictory to the retrieved contexts. We introduce LYNX, a SOTA hallucination detection LLM that is capable of advanced reasoning on challenging real-world hallucination scenarios. To evaluate LYNX, we present HaluBench, a comprehensive hallucination evaluation benchmark, consisting of 15k samples sourced from various real-world domains. Our experiment results show that LYNX outperforms GPT-4o, Claude-3-Sonnet, and"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2407.08488","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.AI","submitted_at":"2024-07-11T13:22:17Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"7aabd81a87c228bc95434a6ff864b087b5cacb081d39f3e761b6b72caed1ca2b","abstract_canon_sha256":"4495629136e4670db211940441a37834a38e3f53990c153e238129a88571e42c"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:47:22.188369Z","signature_b64":"p53v5RUBgYiBkAzjTeF2OnLnXrMYIyN4oR4OxaAMwj2OAjPZEVt6fsVwNoV7YJPrXCM9n1N7rz/x3hLbuQXODA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ba674a44e7c0c1404ed3e48058423649e60088818993ebab9995a1576da94857","last_reissued_at":"2026-07-05T08:47:22.187791Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:47:22.187791Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Lynx: An Open Source Hallucination Evaluation Model","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.AI","authors_text":"Anand Kannappan, Bartosz Mielczarek, Douwe Kiela, Rebecca Qian, Selvan Sunitha Ravi","submitted_at":"2024-07-11T13:22:17Z","abstract_excerpt":"Retrieval Augmented Generation (RAG) techniques aim to mitigate hallucinations in Large Language Models (LLMs). However, LLMs can still produce information that is unsupported or contradictory to the retrieved contexts. We introduce LYNX, a SOTA hallucination detection LLM that is capable of advanced reasoning on challenging real-world hallucination scenarios. To evaluate LYNX, we present HaluBench, a comprehensive hallucination evaluation benchmark, consisting of 15k samples sourced from various real-world domains. Our experiment results show that LYNX outperforms GPT-4o, Claude-3-Sonnet, and"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2407.08488","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2407.08488/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2407.08488","created_at":"2026-07-05T08:47:22.187855+00:00"},{"alias_kind":"arxiv_version","alias_value":"2407.08488v2","created_at":"2026-07-05T08:47:22.187855+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2407.08488","created_at":"2026-07-05T08:47:22.187855+00:00"},{"alias_kind":"pith_short_12","alias_value":"XJTUURHHYDAU","created_at":"2026-07-05T08:47:22.187855+00:00"},{"alias_kind":"pith_short_16","alias_value":"XJTUURHHYDAUATWT","created_at":"2026-07-05T08:47:22.187855+00:00"},{"alias_kind":"pith_short_8","alias_value":"XJTUURHH","created_at":"2026-07-05T08:47:22.187855+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":9,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.18021","citing_title":"LegalHalluLens: Typed Hallucination Auditing and Calibrated Multi-Agent Debate for Trustworthy Legal AI","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2606.08705","citing_title":"Analyzing the Correlation Between Hallucinations and Knowledge Conflicts in Large Language Models","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2605.25376","citing_title":"KYA: A Framework-Agnostic Trust Layer for Autonomous Systems with Verifiable Provenance and Hierarchical Policy Composition","ref_index":70,"is_internal_anchor":false},{"citing_arxiv_id":"2504.00446","citing_title":"Exposing the Ghost in the Transformer: Abnormal Detection for Large Language Models via Hidden State Forensics","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2605.19940","citing_title":"Robotics-Inspired Guardrails for Foundation Models in Socially Sensitive Domains","ref_index":41,"is_internal_anchor":false},{"citing_arxiv_id":"2309.01219","citing_title":"Siren's Song in the AI Ocean: A Survey on Hallucination in Large Language Models","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2605.03476","citing_title":"CuraView: A Multi-Agent Framework for Medical Hallucination Detection with GraphRAG-Enhanced Knowledge Verification","ref_index":46,"is_internal_anchor":false},{"citing_arxiv_id":"2604.23588","citing_title":"FinGround: Detecting and Grounding Financial Hallucinations via Atomic Claim Verification","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2604.15945","citing_title":"RAGognizer: Hallucination-Aware Fine-Tuning via Detection Head Integration","ref_index":22,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/XJTUURHHYDAUATWT4SAFQQRWJH","json":"https://pith.science/pith/XJTUURHHYDAUATWT4SAFQQRWJH.json","graph_json":"https://pith.science/api/pith-number/XJTUURHHYDAUATWT4SAFQQRWJH/graph.json","events_json":"https://pith.science/api/pith-number/XJTUURHHYDAUATWT4SAFQQRWJH/events.json","paper":"https://pith.science/paper/XJTUURHH"},"agent_actions":{"view_html":"https://pith.science/pith/XJTUURHHYDAUATWT4SAFQQRWJH","download_json":"https://pith.science/pith/XJTUURHHYDAUATWT4SAFQQRWJH.json","view_paper":"https://pith.science/paper/XJTUURHH","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2407.08488&json=true","fetch_graph":"https://pith.science/api/pith-number/XJTUURHHYDAUATWT4SAFQQRWJH/graph.json","fetch_events":"https://pith.science/api/pith-number/XJTUURHHYDAUATWT4SAFQQRWJH/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/XJTUURHHYDAUATWT4SAFQQRWJH/action/timestamp_anchor","attest_storage":"https://pith.science/pith/XJTUURHHYDAUATWT4SAFQQRWJH/action/storage_attestation","attest_author":"https://pith.science/pith/XJTUURHHYDAUATWT4SAFQQRWJH/action/author_attestation","sign_citation":"https://pith.science/pith/XJTUURHHYDAUATWT4SAFQQRWJH/action/citation_signature","submit_replication":"https://pith.science/pith/XJTUURHHYDAUATWT4SAFQQRWJH/action/replication_record"}},"created_at":"2026-07-05T08:47:22.187855+00:00","updated_at":"2026-07-05T08:47:22.187855+00:00"}