{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:WN4NYAUMHMAXSTCTR65NGNGPVR","short_pith_number":"pith:WN4NYAUM","schema_version":"1.0","canonical_sha256":"b378dc028c3b01794c538fbad334cfac60b693196ed9a7efe3510f2fc7aa692c","source":{"kind":"arxiv","id":"2502.01419","version":2},"attestation_state":"computed","paper":{"title":"Visual Attention Never Fades: Selective Progressive Attention ReCalibration for Detailed Image Captioning in Multimodal Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Eunji Kim, Mingi Jung, Saehyung Lee, Sungroh Yoon","submitted_at":"2025-02-03T14:58:11Z","abstract_excerpt":"Detailed image captioning is essential for tasks like data generation and aiding visually impaired individuals. High-quality captions require a balance between precision and recall, which remains challenging for current multimodal large language models (MLLMs). In this work, we hypothesize that this limitation stems from weakening and increasingly noisy visual attention as responses lengthen. To address this issue, we propose SPARC (Selective Progressive Attention ReCalibration), a training-free method that enhances the contribution of visual tokens during decoding. SPARC is founded on three k"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.01419","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2025-02-03T14:58:11Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"f5605280e79c7f72bcf0ec540d408c8c73c461941637e1ade066e863f815526e","abstract_canon_sha256":"cd4ede834370d6cb78cb16237021a691c2647a42599eb18e0f0fe9e13e355e31"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:15:42.475090Z","signature_b64":"Zg4BXvsGDTEMuA7maUL2MKLYpf3kEZVS4h+A5cVEqmaLrHxkLAhdqjFO4wony6Q3rikt+l2Y8hbmjLg0tuLZDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b378dc028c3b01794c538fbad334cfac60b693196ed9a7efe3510f2fc7aa692c","last_reissued_at":"2026-07-05T11:15:42.474672Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:15:42.474672Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Visual Attention Never Fades: Selective Progressive Attention ReCalibration for Detailed Image Captioning in Multimodal Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Eunji Kim, Mingi Jung, Saehyung Lee, Sungroh Yoon","submitted_at":"2025-02-03T14:58:11Z","abstract_excerpt":"Detailed image captioning is essential for tasks like data generation and aiding visually impaired individuals. High-quality captions require a balance between precision and recall, which remains challenging for current multimodal large language models (MLLMs). In this work, we hypothesize that this limitation stems from weakening and increasingly noisy visual attention as responses lengthen. To address this issue, we propose SPARC (Selective Progressive Attention ReCalibration), a training-free method that enhances the contribution of visual tokens during decoding. SPARC is founded on three k"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.01419","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.01419/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.01419","created_at":"2026-07-05T11:15:42.474723+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.01419v2","created_at":"2026-07-05T11:15:42.474723+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.01419","created_at":"2026-07-05T11:15:42.474723+00:00"},{"alias_kind":"pith_short_12","alias_value":"WN4NYAUMHMAX","created_at":"2026-07-05T11:15:42.474723+00:00"},{"alias_kind":"pith_short_16","alias_value":"WN4NYAUMHMAXSTCT","created_at":"2026-07-05T11:15:42.474723+00:00"},{"alias_kind":"pith_short_8","alias_value":"WN4NYAUM","created_at":"2026-07-05T11:15:42.474723+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":8,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.25432","citing_title":"Brevity is the Soul of Inference Efficiency: Inducing Concision in VLMs via Data Curation","ref_index":44,"is_internal_anchor":false},{"citing_arxiv_id":"2606.07647","citing_title":"Steer Where It Matters: Token-Level Visual-Sensitivity Steering for LVLMs Hallucination Mitigation","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2606.25432","citing_title":"Brevity is the Soul of Inference Efficiency: Inducing Concision in VLMs via Data Curation","ref_index":44,"is_internal_anchor":false},{"citing_arxiv_id":"2605.25799","citing_title":"Addressing Exacerbated Attention Sink for Source-Free Cross-Domain Few-Shot Learning","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2509.23322","citing_title":"Mitigating Visual Context Degradation in Large Multimodal Models: A Training-Free Decoupled Agentic Framework","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2604.25273","citing_title":"Combating Visual Neglect and Semantic Drift in Large Multimodal Models for Enhanced Cross-Modal Retrieval","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2604.10999","citing_title":"TraversalBench: Challenging Paths to Follow for Vision Language Models","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2604.05497","citing_title":"Thinking Diffusion: Penalize and Guide Visual-Grounded Reasoning in Diffusion Multimodal Language Models","ref_index":13,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/WN4NYAUMHMAXSTCTR65NGNGPVR","json":"https://pith.science/pith/WN4NYAUMHMAXSTCTR65NGNGPVR.json","graph_json":"https://pith.science/api/pith-number/WN4NYAUMHMAXSTCTR65NGNGPVR/graph.json","events_json":"https://pith.science/api/pith-number/WN4NYAUMHMAXSTCTR65NGNGPVR/events.json","paper":"https://pith.science/paper/WN4NYAUM"},"agent_actions":{"view_html":"https://pith.science/pith/WN4NYAUMHMAXSTCTR65NGNGPVR","download_json":"https://pith.science/pith/WN4NYAUMHMAXSTCTR65NGNGPVR.json","view_paper":"https://pith.science/paper/WN4NYAUM","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.01419&json=true","fetch_graph":"https://pith.science/api/pith-number/WN4NYAUMHMAXSTCTR65NGNGPVR/graph.json","fetch_events":"https://pith.science/api/pith-number/WN4NYAUMHMAXSTCTR65NGNGPVR/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/WN4NYAUMHMAXSTCTR65NGNGPVR/action/timestamp_anchor","attest_storage":"https://pith.science/pith/WN4NYAUMHMAXSTCTR65NGNGPVR/action/storage_attestation","attest_author":"https://pith.science/pith/WN4NYAUMHMAXSTCTR65NGNGPVR/action/author_attestation","sign_citation":"https://pith.science/pith/WN4NYAUMHMAXSTCTR65NGNGPVR/action/citation_signature","submit_replication":"https://pith.science/pith/WN4NYAUMHMAXSTCTR65NGNGPVR/action/replication_record"}},"created_at":"2026-07-05T11:15:42.474723+00:00","updated_at":"2026-07-05T11:15:42.474723+00:00"}