{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:FIWHCX2RFD73EHZ446D447WED3","short_pith_number":"pith:FIWHCX2R","schema_version":"1.0","canonical_sha256":"2a2c715f5128ffb21f3ce787ce7ec41ee65fd1f2bbb893e0a6729617427be48c","source":{"kind":"arxiv","id":"2502.03628","version":2},"attestation_state":"computed","paper":{"title":"The Hidden Life of Tokens: Reducing Hallucination of Large Vision-Language Models via Visual Information Steering","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CV","authors_text":"Di Liu, Dimitris N. Metaxas, Haizhou Shi, Hao Wang, Long Zhao, Ting Liu, Yunhe Gao, Yuxiao Chen, Zhenting Wang, Zhuowei Li","submitted_at":"2025-02-05T21:34:02Z","abstract_excerpt":"Large Vision-Language Models (LVLMs) can reason effectively over both textual and visual inputs, but they tend to hallucinate syntactically coherent yet visually ungrounded contents. In this paper, we investigate the internal dynamics of hallucination by examining the tokens logits ranking throughout the generation process, revealing three key patterns in how LVLMs process information: (1) gradual visual information loss - visually grounded tokens gradually become less favored throughout generation, and (2) early excitation - semantically meaningful tokens achieve peak activation in the layers"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.03628","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2025-02-05T21:34:02Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"b4d0d2749fefa4ebe929755be6e06083ccfb8a7d1e5078e387c574fe9f8d00d3","abstract_canon_sha256":"d5b26e9ddad313d80db28f0db5981e00ba8527f51608210a7bf3f953cb4666ae"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:29:54.957244Z","signature_b64":"8riZYND9Pyg9M380IyHB4iVe3hKOUYvnbsqVmaEIQ3FNlJH6050aWscXBf8BAE5sCFrVOCJg4eFntIF+/1XpAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"2a2c715f5128ffb21f3ce787ce7ec41ee65fd1f2bbb893e0a6729617427be48c","last_reissued_at":"2026-07-05T11:29:54.956729Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:29:54.956729Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"The Hidden Life of Tokens: Reducing Hallucination of Large Vision-Language Models via Visual Information Steering","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CV","authors_text":"Di Liu, Dimitris N. Metaxas, Haizhou Shi, Hao Wang, Long Zhao, Ting Liu, Yunhe Gao, Yuxiao Chen, Zhenting Wang, Zhuowei Li","submitted_at":"2025-02-05T21:34:02Z","abstract_excerpt":"Large Vision-Language Models (LVLMs) can reason effectively over both textual and visual inputs, but they tend to hallucinate syntactically coherent yet visually ungrounded contents. In this paper, we investigate the internal dynamics of hallucination by examining the tokens logits ranking throughout the generation process, revealing three key patterns in how LVLMs process information: (1) gradual visual information loss - visually grounded tokens gradually become less favored throughout generation, and (2) early excitation - semantically meaningful tokens achieve peak activation in the layers"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.03628","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.03628/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.03628","created_at":"2026-07-05T11:29:54.956796+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.03628v2","created_at":"2026-07-05T11:29:54.956796+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.03628","created_at":"2026-07-05T11:29:54.956796+00:00"},{"alias_kind":"pith_short_12","alias_value":"FIWHCX2RFD73","created_at":"2026-07-05T11:29:54.956796+00:00"},{"alias_kind":"pith_short_16","alias_value":"FIWHCX2RFD73EHZ4","created_at":"2026-07-05T11:29:54.956796+00:00"},{"alias_kind":"pith_short_8","alias_value":"FIWHCX2R","created_at":"2026-07-05T11:29:54.956796+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":17,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.11792","citing_title":"MultiToP: Learning to Patch Visual Tokens to Mitigate Hallucinations in Video Large Multimodal Models","ref_index":36,"is_internal_anchor":false},{"citing_arxiv_id":"2606.29431","citing_title":"FADE: Mitigating Hallucinations by Reducing Language-Prior Dominance in Large Vision-Language Models","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2606.07647","citing_title":"Steer Where It Matters: Token-Level Visual-Sensitivity Steering for LVLMs Hallucination Mitigation","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01733","citing_title":"GEASS: Gated Evidence-Adaptive Selective Caption Trust for Vision-Language Models","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08245","citing_title":"When Language Overwrites Vision: Over-Alignment and Geometric Debiasing in Vision-Language Models","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2605.24957","citing_title":"Mitigating Object Hallucinations in Vision-Language Models through Region-Aware Attention Recalibration","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2606.29431","citing_title":"FADE: Mitigating Hallucinations by Reducing Language-Prior Dominance in Large Vision-Language Models","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2511.10292","citing_title":"Adaptive Residual-Update Steering for Low-Overhead Hallucination Mitigation in Large Vision Language Models","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01733","citing_title":"GEASS: Gated Evidence-Adaptive Selective Caption Trust for Vision-Language Models","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08245","citing_title":"When Language Overwrites Vision: Over-Alignment and Geometric Debiasing in Vision-Language Models","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08245","citing_title":"When Language Overwrites Vision: Over-Alignment and Geometric Debiasing in Vision-Language Models","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10676","citing_title":"Not Blind but Silenced: Rebalancing Vision and Language via Adversarial Counter-Commonsense Equilibrium","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08245","citing_title":"When Language Overwrites Vision: Over-Alignment and Geometric Debiasing in Vision-Language Models","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2604.25642","citing_title":"Prefill-Time Intervention for Mitigating Hallucination in Large Vision-Language Models","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2404.18930","citing_title":"Hallucination of Multimodal Large Language Models: A Survey","ref_index":108,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01733","citing_title":"GEASS: Gated Evidence-Adaptive Selective Caption Trust for Vision-Language Models","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2604.14090","citing_title":"From Weights to Activations: Is Steering the Next Frontier of Adaptation?","ref_index":11,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/FIWHCX2RFD73EHZ446D447WED3","json":"https://pith.science/pith/FIWHCX2RFD73EHZ446D447WED3.json","graph_json":"https://pith.science/api/pith-number/FIWHCX2RFD73EHZ446D447WED3/graph.json","events_json":"https://pith.science/api/pith-number/FIWHCX2RFD73EHZ446D447WED3/events.json","paper":"https://pith.science/paper/FIWHCX2R"},"agent_actions":{"view_html":"https://pith.science/pith/FIWHCX2RFD73EHZ446D447WED3","download_json":"https://pith.science/pith/FIWHCX2RFD73EHZ446D447WED3.json","view_paper":"https://pith.science/paper/FIWHCX2R","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.03628&json=true","fetch_graph":"https://pith.science/api/pith-number/FIWHCX2RFD73EHZ446D447WED3/graph.json","fetch_events":"https://pith.science/api/pith-number/FIWHCX2RFD73EHZ446D447WED3/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/FIWHCX2RFD73EHZ446D447WED3/action/timestamp_anchor","attest_storage":"https://pith.science/pith/FIWHCX2RFD73EHZ446D447WED3/action/storage_attestation","attest_author":"https://pith.science/pith/FIWHCX2RFD73EHZ446D447WED3/action/author_attestation","sign_citation":"https://pith.science/pith/FIWHCX2RFD73EHZ446D447WED3/action/citation_signature","submit_replication":"https://pith.science/pith/FIWHCX2RFD73EHZ446D447WED3/action/replication_record"}},"created_at":"2026-07-05T11:29:54.956796+00:00","updated_at":"2026-07-05T11:29:54.956796+00:00"}