{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:GEEWP6ZXHQ74ADT6Z75AZIEJOR","short_pith_number":"pith:GEEWP6ZX","schema_version":"1.0","canonical_sha256":"310967fb373c3fc00e7ecffa0ca089747f4015f88aec22d8ce295b3c4cdc79b3","source":{"kind":"arxiv","id":"2402.01345","version":6},"attestation_state":"computed","paper":{"title":"Skip \\n: A Simple Method to Reduce Hallucination in Large Vision-Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.LG"],"primary_cat":"cs.CV","authors_text":"Changqing Zhang, Haiyang Mei, Mike Zheng Shou, Qianli Xu, Zechen Bai, Zongbo Han","submitted_at":"2024-02-02T12:02:46Z","abstract_excerpt":"Recent advancements in large vision-language models (LVLMs) have demonstrated impressive capability in visual information understanding with human language. Despite these advances, LVLMs still face challenges with multimodal hallucination, such as generating text descriptions of objects that are not present in the visual information. However, the underlying fundamental reasons of multimodal hallucinations remain poorly explored. In this paper, we propose a new perspective, suggesting that the inherent biases in LVLMs might be a key factor in hallucinations. Specifically, we systematically iden"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2402.01345","kind":"arxiv","version":6},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2024-02-02T12:02:46Z","cross_cats_sorted":["cs.AI","cs.CL","cs.LG"],"title_canon_sha256":"c7fbae639472487d90856c09738aabb50ac025fb0a389adea3413ca16e489dfd","abstract_canon_sha256":"e1298c5443e66acc132250ff2b73734f3d90c823623d498bbe05f5003db3b039"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:19:08.409439Z","signature_b64":"5TCF/DM2EgLD+jpkpvcbnJWtfOYYsRYDrsHGGCIqGC/CawUlbUzmA1PkWXm2K8Ch8wmjVneVaPvStv4R07XpBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"310967fb373c3fc00e7ecffa0ca089747f4015f88aec22d8ce295b3c4cdc79b3","last_reissued_at":"2026-07-05T08:19:08.408923Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:19:08.408923Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Skip \\n: A Simple Method to Reduce Hallucination in Large Vision-Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.LG"],"primary_cat":"cs.CV","authors_text":"Changqing Zhang, Haiyang Mei, Mike Zheng Shou, Qianli Xu, Zechen Bai, Zongbo Han","submitted_at":"2024-02-02T12:02:46Z","abstract_excerpt":"Recent advancements in large vision-language models (LVLMs) have demonstrated impressive capability in visual information understanding with human language. Despite these advances, LVLMs still face challenges with multimodal hallucination, such as generating text descriptions of objects that are not present in the visual information. However, the underlying fundamental reasons of multimodal hallucinations remain poorly explored. In this paper, we propose a new perspective, suggesting that the inherent biases in LVLMs might be a key factor in hallucinations. Specifically, we systematically iden"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2402.01345","kind":"arxiv","version":6},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2402.01345/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2402.01345","created_at":"2026-07-05T08:19:08.408978+00:00"},{"alias_kind":"arxiv_version","alias_value":"2402.01345v6","created_at":"2026-07-05T08:19:08.408978+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2402.01345","created_at":"2026-07-05T08:19:08.408978+00:00"},{"alias_kind":"pith_short_12","alias_value":"GEEWP6ZXHQ74","created_at":"2026-07-05T08:19:08.408978+00:00"},{"alias_kind":"pith_short_16","alias_value":"GEEWP6ZXHQ74ADT6","created_at":"2026-07-05T08:19:08.408978+00:00"},{"alias_kind":"pith_short_8","alias_value":"GEEWP6ZX","created_at":"2026-07-05T08:19:08.408978+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2507.12455","citing_title":"Mitigating Object Hallucinations via Sentence-Level Early Intervention","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2605.15300","citing_title":"Deep Pre-Alignment for VLMs","ref_index":111,"is_internal_anchor":false},{"citing_arxiv_id":"2404.18930","citing_title":"Hallucination of Multimodal Large Language Models: A Survey","ref_index":59,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/GEEWP6ZXHQ74ADT6Z75AZIEJOR","json":"https://pith.science/pith/GEEWP6ZXHQ74ADT6Z75AZIEJOR.json","graph_json":"https://pith.science/api/pith-number/GEEWP6ZXHQ74ADT6Z75AZIEJOR/graph.json","events_json":"https://pith.science/api/pith-number/GEEWP6ZXHQ74ADT6Z75AZIEJOR/events.json","paper":"https://pith.science/paper/GEEWP6ZX"},"agent_actions":{"view_html":"https://pith.science/pith/GEEWP6ZXHQ74ADT6Z75AZIEJOR","download_json":"https://pith.science/pith/GEEWP6ZXHQ74ADT6Z75AZIEJOR.json","view_paper":"https://pith.science/paper/GEEWP6ZX","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2402.01345&json=true","fetch_graph":"https://pith.science/api/pith-number/GEEWP6ZXHQ74ADT6Z75AZIEJOR/graph.json","fetch_events":"https://pith.science/api/pith-number/GEEWP6ZXHQ74ADT6Z75AZIEJOR/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/GEEWP6ZXHQ74ADT6Z75AZIEJOR/action/timestamp_anchor","attest_storage":"https://pith.science/pith/GEEWP6ZXHQ74ADT6Z75AZIEJOR/action/storage_attestation","attest_author":"https://pith.science/pith/GEEWP6ZXHQ74ADT6Z75AZIEJOR/action/author_attestation","sign_citation":"https://pith.science/pith/GEEWP6ZXHQ74ADT6Z75AZIEJOR/action/citation_signature","submit_replication":"https://pith.science/pith/GEEWP6ZXHQ74ADT6Z75AZIEJOR/action/replication_record"}},"created_at":"2026-07-05T08:19:08.408978+00:00","updated_at":"2026-07-05T08:19:08.408978+00:00"}