{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:CRAHHTUFKFCO4M2SBBCSYY34G6","short_pith_number":"pith:CRAHHTUF","schema_version":"1.0","canonical_sha256":"144073ce855144ee335208452c637c37907e5f4ff31e0b66a8bba572860caeee","source":{"kind":"arxiv","id":"2503.19622","version":1},"attestation_state":"computed","paper":{"title":"Exploring Hallucination of Large Multimodal Models in Video Understanding: Benchmark, Analysis and Mitigation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Baolong Bi, Hongcheng Gao, Hongyu Chen, Jiashu Qu, Jingyi Tang, Li Liang, Li Su, Qingming Huang, Yue Liu","submitted_at":"2025-03-25T13:12:17Z","abstract_excerpt":"The hallucination of large multimodal models (LMMs), providing responses that appear correct but are actually incorrect, limits their reliability and applicability. This paper aims to study the hallucination problem of LMMs in video modality, which is dynamic and more challenging compared to static modalities like images and text. From this motivation, we first present a comprehensive benchmark termed HAVEN for evaluating hallucinations of LMMs in video understanding tasks. It is built upon three dimensions, i.e., hallucination causes, hallucination aspects, and question formats, resulting in "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2503.19622","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2025-03-25T13:12:17Z","cross_cats_sorted":[],"title_canon_sha256":"d34187ee8f60a6f0e245bac8d5339b48866e772cdd82bf673fc6b857b931a842","abstract_canon_sha256":"85695ce585a2846865d362f48b9ddfe7314f72f34f71bb9bdd7db1ef55ce2379"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:38:58.533220Z","signature_b64":"pSTzpJXfvbD4mooyL+AFc6lEqXMzuD0w0ZDeCnez3y+ioUTT9igeqEiNzKyq/DIKgfJdDSzB14uKXwCpKRYiDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"144073ce855144ee335208452c637c37907e5f4ff31e0b66a8bba572860caeee","last_reissued_at":"2026-07-05T10:38:58.532730Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:38:58.532730Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Exploring Hallucination of Large Multimodal Models in Video Understanding: Benchmark, Analysis and Mitigation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Baolong Bi, Hongcheng Gao, Hongyu Chen, Jiashu Qu, Jingyi Tang, Li Liang, Li Su, Qingming Huang, Yue Liu","submitted_at":"2025-03-25T13:12:17Z","abstract_excerpt":"The hallucination of large multimodal models (LMMs), providing responses that appear correct but are actually incorrect, limits their reliability and applicability. This paper aims to study the hallucination problem of LMMs in video modality, which is dynamic and more challenging compared to static modalities like images and text. From this motivation, we first present a comprehensive benchmark termed HAVEN for evaluating hallucinations of LMMs in video understanding tasks. It is built upon three dimensions, i.e., hallucination causes, hallucination aspects, and question formats, resulting in "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2503.19622","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2503.19622/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2503.19622","created_at":"2026-07-05T10:38:58.532791+00:00"},{"alias_kind":"arxiv_version","alias_value":"2503.19622v1","created_at":"2026-07-05T10:38:58.532791+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2503.19622","created_at":"2026-07-05T10:38:58.532791+00:00"},{"alias_kind":"pith_short_12","alias_value":"CRAHHTUFKFCO","created_at":"2026-07-05T10:38:58.532791+00:00"},{"alias_kind":"pith_short_16","alias_value":"CRAHHTUFKFCO4M2S","created_at":"2026-07-05T10:38:58.532791+00:00"},{"alias_kind":"pith_short_8","alias_value":"CRAHHTUF","created_at":"2026-07-05T10:38:58.532791+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":8,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.23061","citing_title":"MotionHalluc: Diagnosing Kinematic Hallucinations in Fine-Grained Motion Reasoning","ref_index":36,"is_internal_anchor":false},{"citing_arxiv_id":"2607.01602","citing_title":"ProWAFT: A ROMA-LPD Instance for Workload-Aware and Dynamic Fault Tolerance in FPGA-Based CNN Accelerators","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2606.11792","citing_title":"MultiToP: Learning to Patch Visual Tokens to Mitigate Hallucinations in Video Large Multimodal Models","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2606.31933","citing_title":"No Place to Hide: Benchmarking Video Hallucination with Background-Controlled Pairs","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2606.30189","citing_title":"DAIN: Dynamic Agent-Based Interaction Network for Efficient and Collaborative Multimodal Reasoning","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2509.08016","citing_title":"Video Parallel Scaling: Aggregating Diverse Frame Subsets for VideoLLMs","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2604.12582","citing_title":"Relaxing Anchor-Frame Dominance for Mitigating Hallucinations in Video Large Language Models","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2604.17375","citing_title":"When Text Hijacks Vision: Benchmarking and Mitigating Text Overlay-Induced Hallucination in Vision Language Models","ref_index":14,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/CRAHHTUFKFCO4M2SBBCSYY34G6","json":"https://pith.science/pith/CRAHHTUFKFCO4M2SBBCSYY34G6.json","graph_json":"https://pith.science/api/pith-number/CRAHHTUFKFCO4M2SBBCSYY34G6/graph.json","events_json":"https://pith.science/api/pith-number/CRAHHTUFKFCO4M2SBBCSYY34G6/events.json","paper":"https://pith.science/paper/CRAHHTUF"},"agent_actions":{"view_html":"https://pith.science/pith/CRAHHTUFKFCO4M2SBBCSYY34G6","download_json":"https://pith.science/pith/CRAHHTUFKFCO4M2SBBCSYY34G6.json","view_paper":"https://pith.science/paper/CRAHHTUF","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2503.19622&json=true","fetch_graph":"https://pith.science/api/pith-number/CRAHHTUFKFCO4M2SBBCSYY34G6/graph.json","fetch_events":"https://pith.science/api/pith-number/CRAHHTUFKFCO4M2SBBCSYY34G6/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/CRAHHTUFKFCO4M2SBBCSYY34G6/action/timestamp_anchor","attest_storage":"https://pith.science/pith/CRAHHTUFKFCO4M2SBBCSYY34G6/action/storage_attestation","attest_author":"https://pith.science/pith/CRAHHTUFKFCO4M2SBBCSYY34G6/action/author_attestation","sign_citation":"https://pith.science/pith/CRAHHTUFKFCO4M2SBBCSYY34G6/action/citation_signature","submit_replication":"https://pith.science/pith/CRAHHTUFKFCO4M2SBBCSYY34G6/action/replication_record"}},"created_at":"2026-07-05T10:38:58.532791+00:00","updated_at":"2026-07-05T10:38:58.532791+00:00"}