{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:CK6ANY7MSAXQHHOIHDVO6DWJXP","short_pith_number":"pith:CK6ANY7M","schema_version":"1.0","canonical_sha256":"12bc06e3ec902f039dc838eaef0ec9bbe6e81c4b45bae6325996b3b71acc5976","source":{"kind":"arxiv","id":"2410.15778","version":2},"attestation_state":"computed","paper":{"title":"Reducing Hallucinations in Vision-Language Models via Latent Space Steering","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG","cs.MM"],"primary_cat":"cs.CV","authors_text":"Haotian Ye, James Zou, Lei Xing, Sheng Liu","submitted_at":"2024-10-21T08:42:30Z","abstract_excerpt":"Hallucination poses a challenge to the deployment of large vision-language models (LVLMs) in applications. Unlike in large language models (LLMs), hallucination in LVLMs often arises from misalignments between visual inputs and textual outputs. This paper investigates the underlying mechanisms of hallucination, focusing on the unique structure of LVLMs that distinguishes them from large language models (LLMs). We identify that hallucinations often arise from the sensitivity of text decoders to vision inputs, a natural phenomenon when image encoders and text decoders are pre-trained separately."},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.15778","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-10-21T08:42:30Z","cross_cats_sorted":["cs.AI","cs.LG","cs.MM"],"title_canon_sha256":"6ef368b469f5dafa931787a3d65ed6cd244a701563bbbe9bfb5825f40d6e2ff9","abstract_canon_sha256":"cb103726e4ddbc9a977f9b62040b93bd0d61dd68b087691e0ee6b694cd06b310"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:24:01.490336Z","signature_b64":"l5DYTZpQqOUBL0ckurHtgz/ApFvZJmVEq2tyqzALueoHVys6B9epegjKLzfkCFwXyqYO23yiWcj/L83abatbAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"12bc06e3ec902f039dc838eaef0ec9bbe6e81c4b45bae6325996b3b71acc5976","last_reissued_at":"2026-07-05T09:24:01.489853Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:24:01.489853Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Reducing Hallucinations in Vision-Language Models via Latent Space Steering","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG","cs.MM"],"primary_cat":"cs.CV","authors_text":"Haotian Ye, James Zou, Lei Xing, Sheng Liu","submitted_at":"2024-10-21T08:42:30Z","abstract_excerpt":"Hallucination poses a challenge to the deployment of large vision-language models (LVLMs) in applications. Unlike in large language models (LLMs), hallucination in LVLMs often arises from misalignments between visual inputs and textual outputs. This paper investigates the underlying mechanisms of hallucination, focusing on the unique structure of LVLMs that distinguishes them from large language models (LLMs). We identify that hallucinations often arise from the sensitivity of text decoders to vision inputs, a natural phenomenon when image encoders and text decoders are pre-trained separately."},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.15778","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.15778/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.15778","created_at":"2026-07-05T09:24:01.489909+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.15778v2","created_at":"2026-07-05T09:24:01.489909+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.15778","created_at":"2026-07-05T09:24:01.489909+00:00"},{"alias_kind":"pith_short_12","alias_value":"CK6ANY7MSAXQ","created_at":"2026-07-05T09:24:01.489909+00:00"},{"alias_kind":"pith_short_16","alias_value":"CK6ANY7MSAXQHHOI","created_at":"2026-07-05T09:24:01.489909+00:00"},{"alias_kind":"pith_short_8","alias_value":"CK6ANY7M","created_at":"2026-07-05T09:24:01.489909+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":7,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.29808","citing_title":"Making Multimodal LLMs Reliable Chart Data Extractors: A Benchmark and Training Framework","ref_index":41,"is_internal_anchor":false},{"citing_arxiv_id":"2604.25642","citing_title":"Prefill-Time Intervention for Mitigating Hallucination in Large Vision-Language Models","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05593","citing_title":"Causal Probing for Internal Visual Representations in Multimodal Large Language Models","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2604.21911","citing_title":"When Prompts Override Vision: Prompt-Induced Hallucinations in LVLMs","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2604.20366","citing_title":"Mitigating Hallucinations in Large Vision-Language Models without Performance Degradation","ref_index":168,"is_internal_anchor":false},{"citing_arxiv_id":"2604.21027","citing_title":"HypEHR: Hyperbolic Modeling of Electronic Health Records for Efficient Question Answering","ref_index":288,"is_internal_anchor":false},{"citing_arxiv_id":"2605.04641","citing_title":"CAST: Mitigating Object Hallucination in Large Vision-Language Models via Caption-Guided Visual Attention Steering","ref_index":80,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/CK6ANY7MSAXQHHOIHDVO6DWJXP","json":"https://pith.science/pith/CK6ANY7MSAXQHHOIHDVO6DWJXP.json","graph_json":"https://pith.science/api/pith-number/CK6ANY7MSAXQHHOIHDVO6DWJXP/graph.json","events_json":"https://pith.science/api/pith-number/CK6ANY7MSAXQHHOIHDVO6DWJXP/events.json","paper":"https://pith.science/paper/CK6ANY7M"},"agent_actions":{"view_html":"https://pith.science/pith/CK6ANY7MSAXQHHOIHDVO6DWJXP","download_json":"https://pith.science/pith/CK6ANY7MSAXQHHOIHDVO6DWJXP.json","view_paper":"https://pith.science/paper/CK6ANY7M","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.15778&json=true","fetch_graph":"https://pith.science/api/pith-number/CK6ANY7MSAXQHHOIHDVO6DWJXP/graph.json","fetch_events":"https://pith.science/api/pith-number/CK6ANY7MSAXQHHOIHDVO6DWJXP/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/CK6ANY7MSAXQHHOIHDVO6DWJXP/action/timestamp_anchor","attest_storage":"https://pith.science/pith/CK6ANY7MSAXQHHOIHDVO6DWJXP/action/storage_attestation","attest_author":"https://pith.science/pith/CK6ANY7MSAXQHHOIHDVO6DWJXP/action/author_attestation","sign_citation":"https://pith.science/pith/CK6ANY7MSAXQHHOIHDVO6DWJXP/action/citation_signature","submit_replication":"https://pith.science/pith/CK6ANY7MSAXQHHOIHDVO6DWJXP/action/replication_record"}},"created_at":"2026-07-05T09:24:01.489909+00:00","updated_at":"2026-07-05T09:24:01.489909+00:00"}