{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:XZ6TPOIP6J6BQTA4DDOHA67ERJ","short_pith_number":"pith:XZ6TPOIP","schema_version":"1.0","canonical_sha256":"be7d37b90ff27c184c1c18dc707be48a46cb1c9facb38ab4a879a6cf1b10378b","source":{"kind":"arxiv","id":"2411.15839","version":2},"attestation_state":"computed","paper":{"title":"VaLiD: Mitigating the Hallucination of Large Vision Language Models by Visual Layer Fusion Contrastive Decoding","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Jiaqi Wang, Jitao Sang, Yifei Gao","submitted_at":"2024-11-24T13:42:02Z","abstract_excerpt":"Large Vision-Language Models (LVLMs) have demonstrated remarkable capabilities in multimodal task reasoning. However, they often generate responses that appear plausible yet do not accurately reflect the visual content, a phenomenon known as hallucination. Recent approaches have introduced training-free methods to mitigate hallucinations by adjusting the decoding strategy during the inference stage, typically attributing hallucinations to the language model itself. Our analysis, however, reveals that distortions in the visual encoding process significantly affect the model's reasoning capabili"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2411.15839","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-11-24T13:42:02Z","cross_cats_sorted":[],"title_canon_sha256":"b32f6a3994ec1a89db6b54294795bf93a49e37194aa9cc304249069dd497ab19","abstract_canon_sha256":"0c150369dc57ce9abed8fa2f2c806f47fe4022ea476109f119718e713eef889d"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:31:50.725408Z","signature_b64":"n0rPyjyswkS2Dcx7oUlvTrC90ANCpp976rFzz6Ho/1WoiOQGvJAQHf++e1ijzePhXaiSdUVlKL2z0u7v9XoOCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"be7d37b90ff27c184c1c18dc707be48a46cb1c9facb38ab4a879a6cf1b10378b","last_reissued_at":"2026-07-05T10:31:50.724735Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:31:50.724735Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"VaLiD: Mitigating the Hallucination of Large Vision Language Models by Visual Layer Fusion Contrastive Decoding","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Jiaqi Wang, Jitao Sang, Yifei Gao","submitted_at":"2024-11-24T13:42:02Z","abstract_excerpt":"Large Vision-Language Models (LVLMs) have demonstrated remarkable capabilities in multimodal task reasoning. However, they often generate responses that appear plausible yet do not accurately reflect the visual content, a phenomenon known as hallucination. Recent approaches have introduced training-free methods to mitigate hallucinations by adjusting the decoding strategy during the inference stage, typically attributing hallucinations to the language model itself. Our analysis, however, reveals that distortions in the visual encoding process significantly affect the model's reasoning capabili"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2411.15839","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2411.15839/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2411.15839","created_at":"2026-07-05T10:31:50.724813+00:00"},{"alias_kind":"arxiv_version","alias_value":"2411.15839v2","created_at":"2026-07-05T10:31:50.724813+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2411.15839","created_at":"2026-07-05T10:31:50.724813+00:00"},{"alias_kind":"pith_short_12","alias_value":"XZ6TPOIP6J6B","created_at":"2026-07-05T10:31:50.724813+00:00"},{"alias_kind":"pith_short_16","alias_value":"XZ6TPOIP6J6BQTA4","created_at":"2026-07-05T10:31:50.724813+00:00"},{"alias_kind":"pith_short_8","alias_value":"XZ6TPOIP","created_at":"2026-07-05T10:31:50.724813+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.25325","citing_title":"Omni-Perception Policy Optimization for Multimodal Emotion Reasoning","ref_index":75,"is_internal_anchor":false},{"citing_arxiv_id":"2606.17953","citing_title":"MLLMs Get It Right, Then Get It Wrong: Tracing and Correcting Late-Layer Textual Bias","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2404.18930","citing_title":"Hallucination of Multimodal Large Language Models: A Survey","ref_index":163,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/XZ6TPOIP6J6BQTA4DDOHA67ERJ","json":"https://pith.science/pith/XZ6TPOIP6J6BQTA4DDOHA67ERJ.json","graph_json":"https://pith.science/api/pith-number/XZ6TPOIP6J6BQTA4DDOHA67ERJ/graph.json","events_json":"https://pith.science/api/pith-number/XZ6TPOIP6J6BQTA4DDOHA67ERJ/events.json","paper":"https://pith.science/paper/XZ6TPOIP"},"agent_actions":{"view_html":"https://pith.science/pith/XZ6TPOIP6J6BQTA4DDOHA67ERJ","download_json":"https://pith.science/pith/XZ6TPOIP6J6BQTA4DDOHA67ERJ.json","view_paper":"https://pith.science/paper/XZ6TPOIP","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2411.15839&json=true","fetch_graph":"https://pith.science/api/pith-number/XZ6TPOIP6J6BQTA4DDOHA67ERJ/graph.json","fetch_events":"https://pith.science/api/pith-number/XZ6TPOIP6J6BQTA4DDOHA67ERJ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/XZ6TPOIP6J6BQTA4DDOHA67ERJ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/XZ6TPOIP6J6BQTA4DDOHA67ERJ/action/storage_attestation","attest_author":"https://pith.science/pith/XZ6TPOIP6J6BQTA4DDOHA67ERJ/action/author_attestation","sign_citation":"https://pith.science/pith/XZ6TPOIP6J6BQTA4DDOHA67ERJ/action/citation_signature","submit_replication":"https://pith.science/pith/XZ6TPOIP6J6BQTA4DDOHA67ERJ/action/replication_record"}},"created_at":"2026-07-05T10:31:50.724813+00:00","updated_at":"2026-07-05T10:31:50.724813+00:00"}