{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:FXNAGOVHPTHD7MH5BXSK7GOIKF","short_pith_number":"pith:FXNAGOVH","schema_version":"1.0","canonical_sha256":"2dda033aa77cce3fb0fd0de4af99c85173a40b68fc3ca08748374ba9ff063420","source":{"kind":"arxiv","id":"2405.17820","version":2},"attestation_state":"computed","paper":{"title":"Don't Miss the Forest for the Trees: Attentional Vision Calibration for Large Vision Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Changick Kim, Donguk Kim, Jaehyuk Jang, Sangmin Woo, Yubin Choi","submitted_at":"2024-05-28T04:40:57Z","abstract_excerpt":"Large Vision Language Models (LVLMs) demonstrate strong capabilities in visual understanding and description, yet often suffer from hallucinations, attributing incorrect or misleading features to images. We observe that LVLMs disproportionately focus on a small subset of image tokens--termed blind tokens--which are typically irrelevant to the query (e.g., background or non-object regions). We hypothesize that such attention misalignment plays a key role in generating hallucinated responses. To mitigate this issue, we propose Attentional Vision Calibration (AvisC), a test-time approach that dyn"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2405.17820","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2024-05-28T04:40:57Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"ea621c7062b2408594c62254740b9711ef376d9a9f4912669b25deb52c0e3f00","abstract_canon_sha256":"b3fc1ea9e00a6919143e68f40a24772181abfcd4c22558cff62136b737286373"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:13:10.089099Z","signature_b64":"/yVYjJXFGrm+p0ICqkuy+oDOCFxlZNcwKZp3aQblcxy36Zx1PCZTe4Pcfqvzi+VeJfcW4N6Y++d28lpbVHhECA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"2dda033aa77cce3fb0fd0de4af99c85173a40b68fc3ca08748374ba9ff063420","last_reissued_at":"2026-07-05T11:13:10.088556Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:13:10.088556Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Don't Miss the Forest for the Trees: Attentional Vision Calibration for Large Vision Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Changick Kim, Donguk Kim, Jaehyuk Jang, Sangmin Woo, Yubin Choi","submitted_at":"2024-05-28T04:40:57Z","abstract_excerpt":"Large Vision Language Models (LVLMs) demonstrate strong capabilities in visual understanding and description, yet often suffer from hallucinations, attributing incorrect or misleading features to images. We observe that LVLMs disproportionately focus on a small subset of image tokens--termed blind tokens--which are typically irrelevant to the query (e.g., background or non-object regions). We hypothesize that such attention misalignment plays a key role in generating hallucinated responses. To mitigate this issue, we propose Attentional Vision Calibration (AvisC), a test-time approach that dyn"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2405.17820","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2405.17820/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2405.17820","created_at":"2026-07-05T11:13:10.088620+00:00"},{"alias_kind":"arxiv_version","alias_value":"2405.17820v2","created_at":"2026-07-05T11:13:10.088620+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2405.17820","created_at":"2026-07-05T11:13:10.088620+00:00"},{"alias_kind":"pith_short_12","alias_value":"FXNAGOVHPTHD","created_at":"2026-07-05T11:13:10.088620+00:00"},{"alias_kind":"pith_short_16","alias_value":"FXNAGOVHPTHD7MH5","created_at":"2026-07-05T11:13:10.088620+00:00"},{"alias_kind":"pith_short_8","alias_value":"FXNAGOVH","created_at":"2026-07-05T11:13:10.088620+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2607.00060","citing_title":"Synergistic Perception-Reasoning Governance: Grounding Medical MLLMs with Verifiable Anatomical Evidence","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2505.21472","citing_title":"Mitigating Hallucination in Large Vision-Language Models via Adaptive Attention Calibration","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2506.09522","citing_title":"Revisit What You See: Revealing Visual Semantics in Vision Tokens to Guide LVLM Decoding","ref_index":46,"is_internal_anchor":false},{"citing_arxiv_id":"2404.18930","citing_title":"Hallucination of Multimodal Large Language Models: A Survey","ref_index":174,"is_internal_anchor":false},{"citing_arxiv_id":"2604.17982","citing_title":"Mitigating Multimodal Hallucination via Phase-wise Self-reward","ref_index":51,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/FXNAGOVHPTHD7MH5BXSK7GOIKF","json":"https://pith.science/pith/FXNAGOVHPTHD7MH5BXSK7GOIKF.json","graph_json":"https://pith.science/api/pith-number/FXNAGOVHPTHD7MH5BXSK7GOIKF/graph.json","events_json":"https://pith.science/api/pith-number/FXNAGOVHPTHD7MH5BXSK7GOIKF/events.json","paper":"https://pith.science/paper/FXNAGOVH"},"agent_actions":{"view_html":"https://pith.science/pith/FXNAGOVHPTHD7MH5BXSK7GOIKF","download_json":"https://pith.science/pith/FXNAGOVHPTHD7MH5BXSK7GOIKF.json","view_paper":"https://pith.science/paper/FXNAGOVH","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2405.17820&json=true","fetch_graph":"https://pith.science/api/pith-number/FXNAGOVHPTHD7MH5BXSK7GOIKF/graph.json","fetch_events":"https://pith.science/api/pith-number/FXNAGOVHPTHD7MH5BXSK7GOIKF/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/FXNAGOVHPTHD7MH5BXSK7GOIKF/action/timestamp_anchor","attest_storage":"https://pith.science/pith/FXNAGOVHPTHD7MH5BXSK7GOIKF/action/storage_attestation","attest_author":"https://pith.science/pith/FXNAGOVHPTHD7MH5BXSK7GOIKF/action/author_attestation","sign_citation":"https://pith.science/pith/FXNAGOVHPTHD7MH5BXSK7GOIKF/action/citation_signature","submit_replication":"https://pith.science/pith/FXNAGOVHPTHD7MH5BXSK7GOIKF/action/replication_record"}},"created_at":"2026-07-05T11:13:10.088620+00:00","updated_at":"2026-07-05T11:13:10.088620+00:00"}