{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:JW2LP6CK6L2FZ63OX6C37PB753","short_pith_number":"pith:JW2LP6CK","schema_version":"1.0","canonical_sha256":"4db4b7f84af2f45cfb6ebf85bfbc3feecba22b2c7f8be4ca4e0d3ad885c0069a","source":{"kind":"arxiv","id":"2406.09155","version":1},"attestation_state":"computed","paper":{"title":"DefAn: Definitive Answer Dataset for LLMs Hallucination Evaluation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CV","cs.LG"],"primary_cat":"cs.CL","authors_text":"A B M Ashikur Rahman, Ajmal Mian, Muhammad Usman, Saeed Anwar","submitted_at":"2024-06-13T14:18:13Z","abstract_excerpt":"Large Language Models (LLMs) have demonstrated remarkable capabilities, revolutionizing the integration of AI in daily life applications. However, they are prone to hallucinations, generating claims that contradict established facts, deviating from prompts, and producing inconsistent responses when the same prompt is presented multiple times. Addressing these issues is challenging due to the lack of comprehensive and easily assessable benchmark datasets. Most existing datasets are small and rely on multiple-choice questions, which are inadequate for evaluating the generative prowess of LLMs. T"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2406.09155","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-06-13T14:18:13Z","cross_cats_sorted":["cs.AI","cs.CV","cs.LG"],"title_canon_sha256":"a0aa75c9bb666da951adacb3ff06c528cb119ba1c5bbe6086c33c854b3180bbc","abstract_canon_sha256":"c9c72adc59aab8c68c74c5bef92fa53986a72e61a2e61a9036689ace65e4d196"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:31:29.115387Z","signature_b64":"oU7llrUCaqDXdaeSwfHdTj6HBuscH7nQ7NtnqSdEt+Pqp/ziBWGA9Yg4d/SrmmWAYRUC/wo10vu0Lr5o/k4XDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"4db4b7f84af2f45cfb6ebf85bfbc3feecba22b2c7f8be4ca4e0d3ad885c0069a","last_reissued_at":"2026-07-05T08:31:29.114861Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:31:29.114861Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"DefAn: Definitive Answer Dataset for LLMs Hallucination Evaluation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CV","cs.LG"],"primary_cat":"cs.CL","authors_text":"A B M Ashikur Rahman, Ajmal Mian, Muhammad Usman, Saeed Anwar","submitted_at":"2024-06-13T14:18:13Z","abstract_excerpt":"Large Language Models (LLMs) have demonstrated remarkable capabilities, revolutionizing the integration of AI in daily life applications. However, they are prone to hallucinations, generating claims that contradict established facts, deviating from prompts, and producing inconsistent responses when the same prompt is presented multiple times. Addressing these issues is challenging due to the lack of comprehensive and easily assessable benchmark datasets. Most existing datasets are small and rely on multiple-choice questions, which are inadequate for evaluating the generative prowess of LLMs. T"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.09155","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.09155/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2406.09155","created_at":"2026-07-05T08:31:29.114927+00:00"},{"alias_kind":"arxiv_version","alias_value":"2406.09155v1","created_at":"2026-07-05T08:31:29.114927+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.09155","created_at":"2026-07-05T08:31:29.114927+00:00"},{"alias_kind":"pith_short_12","alias_value":"JW2LP6CK6L2F","created_at":"2026-07-05T08:31:29.114927+00:00"},{"alias_kind":"pith_short_16","alias_value":"JW2LP6CK6L2FZ63O","created_at":"2026-07-05T08:31:29.114927+00:00"},{"alias_kind":"pith_short_8","alias_value":"JW2LP6CK","created_at":"2026-07-05T08:31:29.114927+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/JW2LP6CK6L2FZ63OX6C37PB753","json":"https://pith.science/pith/JW2LP6CK6L2FZ63OX6C37PB753.json","graph_json":"https://pith.science/api/pith-number/JW2LP6CK6L2FZ63OX6C37PB753/graph.json","events_json":"https://pith.science/api/pith-number/JW2LP6CK6L2FZ63OX6C37PB753/events.json","paper":"https://pith.science/paper/JW2LP6CK"},"agent_actions":{"view_html":"https://pith.science/pith/JW2LP6CK6L2FZ63OX6C37PB753","download_json":"https://pith.science/pith/JW2LP6CK6L2FZ63OX6C37PB753.json","view_paper":"https://pith.science/paper/JW2LP6CK","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2406.09155&json=true","fetch_graph":"https://pith.science/api/pith-number/JW2LP6CK6L2FZ63OX6C37PB753/graph.json","fetch_events":"https://pith.science/api/pith-number/JW2LP6CK6L2FZ63OX6C37PB753/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/JW2LP6CK6L2FZ63OX6C37PB753/action/timestamp_anchor","attest_storage":"https://pith.science/pith/JW2LP6CK6L2FZ63OX6C37PB753/action/storage_attestation","attest_author":"https://pith.science/pith/JW2LP6CK6L2FZ63OX6C37PB753/action/author_attestation","sign_citation":"https://pith.science/pith/JW2LP6CK6L2FZ63OX6C37PB753/action/citation_signature","submit_replication":"https://pith.science/pith/JW2LP6CK6L2FZ63OX6C37PB753/action/replication_record"}},"created_at":"2026-07-05T08:31:29.114927+00:00","updated_at":"2026-07-05T08:31:29.114927+00:00"}