{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:T3S5HTVD5PKMIPTVAUHCHRELU6","short_pith_number":"pith:T3S5HTVD","schema_version":"1.0","canonical_sha256":"9ee5d3cea3ebd4c43e75050e23c48ba7abde53dc8cabbadc92859679f3acc372","source":{"kind":"arxiv","id":"2406.16338","version":1},"attestation_state":"computed","paper":{"title":"VideoHallucer: Evaluating Intrinsic and Extrinsic Hallucinations in Large Video-Language Models","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Cihang Xie, Dongyan Zhao, Yueqian Wang, Yuxuan Wang, Zilong Zheng","submitted_at":"2024-06-24T06:21:59Z","abstract_excerpt":"Recent advancements in Multimodal Large Language Models (MLLMs) have extended their capabilities to video understanding. Yet, these models are often plagued by \"hallucinations\", where irrelevant or nonsensical content is generated, deviating from the actual video context. This work introduces VideoHallucer, the first comprehensive benchmark for hallucination detection in large video-language models (LVLMs). VideoHallucer categorizes hallucinations into two main types: intrinsic and extrinsic, offering further subcategories for detailed analysis, including object-relation, temporal, semantic de"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2406.16338","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.CV","submitted_at":"2024-06-24T06:21:59Z","cross_cats_sorted":[],"title_canon_sha256":"3e319619e2618d048061a6847efa0f1afdbc90f11dbac596bf48707575aaab7b","abstract_canon_sha256":"96750b680b6e0eab553737d25fc5c966e172d6a6e83430731bdcfb43b0247763"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:35:56.966391Z","signature_b64":"fJcZa72Smxvrbreq9ftwFrIDOb5RIf8jUQXxaStyjX12gJx/lflpTRN6dWPbTXsr/40BVu1r6iCGbd4FDezMDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"9ee5d3cea3ebd4c43e75050e23c48ba7abde53dc8cabbadc92859679f3acc372","last_reissued_at":"2026-07-05T08:35:56.966026Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:35:56.966026Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"VideoHallucer: Evaluating Intrinsic and Extrinsic Hallucinations in Large Video-Language Models","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Cihang Xie, Dongyan Zhao, Yueqian Wang, Yuxuan Wang, Zilong Zheng","submitted_at":"2024-06-24T06:21:59Z","abstract_excerpt":"Recent advancements in Multimodal Large Language Models (MLLMs) have extended their capabilities to video understanding. Yet, these models are often plagued by \"hallucinations\", where irrelevant or nonsensical content is generated, deviating from the actual video context. This work introduces VideoHallucer, the first comprehensive benchmark for hallucination detection in large video-language models (LVLMs). VideoHallucer categorizes hallucinations into two main types: intrinsic and extrinsic, offering further subcategories for detailed analysis, including object-relation, temporal, semantic de"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.16338","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.16338/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2406.16338","created_at":"2026-07-05T08:35:56.966088+00:00"},{"alias_kind":"arxiv_version","alias_value":"2406.16338v1","created_at":"2026-07-05T08:35:56.966088+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.16338","created_at":"2026-07-05T08:35:56.966088+00:00"},{"alias_kind":"pith_short_12","alias_value":"T3S5HTVD5PKM","created_at":"2026-07-05T08:35:56.966088+00:00"},{"alias_kind":"pith_short_16","alias_value":"T3S5HTVD5PKMIPTV","created_at":"2026-07-05T08:35:56.966088+00:00"},{"alias_kind":"pith_short_8","alias_value":"T3S5HTVD","created_at":"2026-07-05T08:35:56.966088+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":13,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.11792","citing_title":"MultiToP: Learning to Patch Visual Tokens to Mitigate Hallucinations in Video Large Multimodal Models","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2607.01117","citing_title":"MoHallBench: A Benchmark for Motion Hallucination in Video Large Language Models","ref_index":38,"is_internal_anchor":false},{"citing_arxiv_id":"2606.03614","citing_title":"OmniHalluc-L: Counterfactual Benchmarking and Modality-Perturbation Reliability Calibration for Long-Form Omni Hallucination","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2606.31933","citing_title":"No Place to Hide: Benchmarking Video Hallucination with Background-Controlled Pairs","ref_index":71,"is_internal_anchor":false},{"citing_arxiv_id":"2411.16771","citing_title":"VidHal: Benchmarking Temporal Hallucinations in Vision LLMs","ref_index":53,"is_internal_anchor":false},{"citing_arxiv_id":"2605.16403","citing_title":"When Vision Speaks for Sound","ref_index":64,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09904","citing_title":"TOC-Bench: A Temporal Object Consistency Benchmark for Video Large Language Models","ref_index":40,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09904","citing_title":"TOC-Bench: A Temporal Object Consistency Benchmark for Video Large Language Models","ref_index":40,"is_internal_anchor":false},{"citing_arxiv_id":"2605.04515","citing_title":"From Priors to Perception: Grounding Video-LLMs in Physical Reality","ref_index":38,"is_internal_anchor":false},{"citing_arxiv_id":"2604.20473","citing_title":"Video-ToC: Video Tree-of-Cue Reasoning","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2604.12582","citing_title":"Relaxing Anchor-Frame Dominance for Mitigating Hallucinations in Video Large Language Models","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2604.17820","citing_title":"Raven: Rethinking Automated Assessment for Scratch Programs via Video-Grounded Evaluation","ref_index":63,"is_internal_anchor":false},{"citing_arxiv_id":"2604.17375","citing_title":"When Text Hijacks Vision: Benchmarking and Mitigating Text Overlay-Induced Hallucination in Vision Language Models","ref_index":62,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/T3S5HTVD5PKMIPTVAUHCHRELU6","json":"https://pith.science/pith/T3S5HTVD5PKMIPTVAUHCHRELU6.json","graph_json":"https://pith.science/api/pith-number/T3S5HTVD5PKMIPTVAUHCHRELU6/graph.json","events_json":"https://pith.science/api/pith-number/T3S5HTVD5PKMIPTVAUHCHRELU6/events.json","paper":"https://pith.science/paper/T3S5HTVD"},"agent_actions":{"view_html":"https://pith.science/pith/T3S5HTVD5PKMIPTVAUHCHRELU6","download_json":"https://pith.science/pith/T3S5HTVD5PKMIPTVAUHCHRELU6.json","view_paper":"https://pith.science/paper/T3S5HTVD","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2406.16338&json=true","fetch_graph":"https://pith.science/api/pith-number/T3S5HTVD5PKMIPTVAUHCHRELU6/graph.json","fetch_events":"https://pith.science/api/pith-number/T3S5HTVD5PKMIPTVAUHCHRELU6/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/T3S5HTVD5PKMIPTVAUHCHRELU6/action/timestamp_anchor","attest_storage":"https://pith.science/pith/T3S5HTVD5PKMIPTVAUHCHRELU6/action/storage_attestation","attest_author":"https://pith.science/pith/T3S5HTVD5PKMIPTVAUHCHRELU6/action/author_attestation","sign_citation":"https://pith.science/pith/T3S5HTVD5PKMIPTVAUHCHRELU6/action/citation_signature","submit_replication":"https://pith.science/pith/T3S5HTVD5PKMIPTVAUHCHRELU6/action/replication_record"}},"created_at":"2026-07-05T08:35:56.966088+00:00","updated_at":"2026-07-05T08:35:56.966088+00:00"}