{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:T73NNZ6JPHKGWR3ZTJ5WX5DPIY","short_pith_number":"pith:T73NNZ6J","schema_version":"1.0","canonical_sha256":"9ff6d6e7c979d46b47799a7b6bf46f462f5e3b52fb7660f3086d4e9301e9e7f0","source":{"kind":"arxiv","id":"2412.20622","version":2},"attestation_state":"computed","paper":{"title":"Towards a Systematic Evaluation of Hallucinations in Large-Vision Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Ashish Seth, Chirag Agarwal, Dinesh Manocha","submitted_at":"2024-12-29T23:56:01Z","abstract_excerpt":"Large Vision-Language Models (LVLMs) have demonstrated remarkable performance in complex multimodal tasks. However, these models still suffer from hallucinations, particularly when required to implicitly recognize or infer diverse visual entities from images for complex vision-language tasks. To address this challenge, we propose HALLUCINOGEN, a novel visual question answering (VQA) benchmark that employs contextual reasoning prompts as hallucination attacks to evaluate the extent of hallucination in state-of-the-art LVLMs. Our benchmark provides a comprehensive study of the implicit reasoning"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2412.20622","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2024-12-29T23:56:01Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"850fa3cc77b03354658afc3605b9929c101172e09ccc8ea6c86ef5f59591c933","abstract_canon_sha256":"f5a35ac318b8ead438d7a9a90f2447aaef3a288f8792c510c2460db6f02466f1"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:30:59.695268Z","signature_b64":"WF2a7tBczLJvzyI8UueekaS1qRuPVqSbhY9IYGNoDiRg1Eo/8Bb7RUzXSCPga8VxUn8PSDATiCxdYwpMQ8MLCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"9ff6d6e7c979d46b47799a7b6bf46f462f5e3b52fb7660f3086d4e9301e9e7f0","last_reissued_at":"2026-07-05T10:30:59.694336Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:30:59.694336Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Towards a Systematic Evaluation of Hallucinations in Large-Vision Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Ashish Seth, Chirag Agarwal, Dinesh Manocha","submitted_at":"2024-12-29T23:56:01Z","abstract_excerpt":"Large Vision-Language Models (LVLMs) have demonstrated remarkable performance in complex multimodal tasks. However, these models still suffer from hallucinations, particularly when required to implicitly recognize or infer diverse visual entities from images for complex vision-language tasks. To address this challenge, we propose HALLUCINOGEN, a novel visual question answering (VQA) benchmark that employs contextual reasoning prompts as hallucination attacks to evaluate the extent of hallucination in state-of-the-art LVLMs. Our benchmark provides a comprehensive study of the implicit reasoning"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2412.20622","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2412.20622/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2412.20622","created_at":"2026-07-05T10:30:59.694451+00:00"},{"alias_kind":"arxiv_version","alias_value":"2412.20622v2","created_at":"2026-07-05T10:30:59.694451+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2412.20622","created_at":"2026-07-05T10:30:59.694451+00:00"},{"alias_kind":"pith_short_12","alias_value":"T73NNZ6JPHKG","created_at":"2026-07-05T10:30:59.694451+00:00"},{"alias_kind":"pith_short_16","alias_value":"T73NNZ6JPHKGWR3Z","created_at":"2026-07-05T10:30:59.694451+00:00"},{"alias_kind":"pith_short_8","alias_value":"T73NNZ6J","created_at":"2026-07-05T10:30:59.694451+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2508.01781","citing_title":"A comprehensive taxonomy of hallucinations in Large Language Models","ref_index":88,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/T73NNZ6JPHKGWR3ZTJ5WX5DPIY","json":"https://pith.science/pith/T73NNZ6JPHKGWR3ZTJ5WX5DPIY.json","graph_json":"https://pith.science/api/pith-number/T73NNZ6JPHKGWR3ZTJ5WX5DPIY/graph.json","events_json":"https://pith.science/api/pith-number/T73NNZ6JPHKGWR3ZTJ5WX5DPIY/events.json","paper":"https://pith.science/paper/T73NNZ6J"},"agent_actions":{"view_html":"https://pith.science/pith/T73NNZ6JPHKGWR3ZTJ5WX5DPIY","download_json":"https://pith.science/pith/T73NNZ6JPHKGWR3ZTJ5WX5DPIY.json","view_paper":"https://pith.science/paper/T73NNZ6J","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2412.20622&json=true","fetch_graph":"https://pith.science/api/pith-number/T73NNZ6JPHKGWR3ZTJ5WX5DPIY/graph.json","fetch_events":"https://pith.science/api/pith-number/T73NNZ6JPHKGWR3ZTJ5WX5DPIY/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/T73NNZ6JPHKGWR3ZTJ5WX5DPIY/action/timestamp_anchor","attest_storage":"https://pith.science/pith/T73NNZ6JPHKGWR3ZTJ5WX5DPIY/action/storage_attestation","attest_author":"https://pith.science/pith/T73NNZ6JPHKGWR3ZTJ5WX5DPIY/action/author_attestation","sign_citation":"https://pith.science/pith/T73NNZ6JPHKGWR3ZTJ5WX5DPIY/action/citation_signature","submit_replication":"https://pith.science/pith/T73NNZ6JPHKGWR3ZTJ5WX5DPIY/action/replication_record"}},"created_at":"2026-07-05T10:30:59.694451+00:00","updated_at":"2026-07-05T10:30:59.694451+00:00"}