{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:PFXRPNKNVTA5QJCODEM3P7AR3H","short_pith_number":"pith:PFXRPNKN","schema_version":"1.0","canonical_sha256":"796f17b54dacc1d8244e1919b7fc11d9fc494dcd01742b715a26d070356ad9d0","source":{"kind":"arxiv","id":"2408.02032","version":3},"attestation_state":"computed","paper":{"title":"Self-Introspective Decoding: Alleviating Hallucinations for Large Vision-Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Fushuo Huo, Haozhao Wang, Peilin Zhao, Wenchao Xu, Zhicheng Chen, Zhong Zhang","submitted_at":"2024-08-04T13:50:17Z","abstract_excerpt":"While Large Vision-Language Models (LVLMs) have rapidly advanced in recent years, the prevalent issue known as the `hallucination' problem has emerged as a significant bottleneck, hindering their real-world deployments. Existing methods mitigate this issue mainly from two perspectives: One approach leverages extra knowledge like robust instruction tuning LVLMs with curated datasets or employing auxiliary analysis networks, which inevitable incur additional costs. Another approach, known as contrastive decoding, induces hallucinations by manually disturbing the vision or instruction raw inputs "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2408.02032","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2024-08-04T13:50:17Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"8101c2323cc7f24a678ee443892a312ff47376a101cd8a5c2e82d453ef706d83","abstract_canon_sha256":"68421c3d47d89eca5f25af858da4d7ab41001c52043d65b7cf728077750329ed"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:31:45.649101Z","signature_b64":"7SSmiy60vWGWbRLGe2dbYyoZRdWbcADQmSjb20Vu5HqewJwFKQvBjLEGrhOnHgwxZ0Hrds/h6y0LFvQQ80NDCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"796f17b54dacc1d8244e1919b7fc11d9fc494dcd01742b715a26d070356ad9d0","last_reissued_at":"2026-07-05T10:31:45.648588Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:31:45.648588Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Self-Introspective Decoding: Alleviating Hallucinations for Large Vision-Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Fushuo Huo, Haozhao Wang, Peilin Zhao, Wenchao Xu, Zhicheng Chen, Zhong Zhang","submitted_at":"2024-08-04T13:50:17Z","abstract_excerpt":"While Large Vision-Language Models (LVLMs) have rapidly advanced in recent years, the prevalent issue known as the `hallucination' problem has emerged as a significant bottleneck, hindering their real-world deployments. Existing methods mitigate this issue mainly from two perspectives: One approach leverages extra knowledge like robust instruction tuning LVLMs with curated datasets or employing auxiliary analysis networks, which inevitable incur additional costs. Another approach, known as contrastive decoding, induces hallucinations by manually disturbing the vision or instruction raw inputs "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2408.02032","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2408.02032/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2408.02032","created_at":"2026-07-05T10:31:45.648640+00:00"},{"alias_kind":"arxiv_version","alias_value":"2408.02032v3","created_at":"2026-07-05T10:31:45.648640+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2408.02032","created_at":"2026-07-05T10:31:45.648640+00:00"},{"alias_kind":"pith_short_12","alias_value":"PFXRPNKNVTA5","created_at":"2026-07-05T10:31:45.648640+00:00"},{"alias_kind":"pith_short_16","alias_value":"PFXRPNKNVTA5QJCO","created_at":"2026-07-05T10:31:45.648640+00:00"},{"alias_kind":"pith_short_8","alias_value":"PFXRPNKN","created_at":"2026-07-05T10:31:45.648640+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":13,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.07647","citing_title":"Steer Where It Matters: Token-Level Visual-Sensitivity Steering for LVLMs Hallucination Mitigation","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2606.27596","citing_title":"Dismantling Pathological Shortcuts: A Causal Framework for Faithful LVLM Decoding","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2606.31054","citing_title":"ADAPT: Attention Dynamics Alignment with Preference Tuning for Faithful MLLMs","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08245","citing_title":"When Language Overwrites Vision: Over-Alignment and Geometric Debiasing in Vision-Language Models","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2606.29847","citing_title":"See Only When Needed: Context-Aware Attention Intervention for Mitigating Hallucinations in LVLMs","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2605.31429","citing_title":"YARD: Y-Architecture Register Decoding for Efficient Hallucination Mitigation in Large Vision-Language Models","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2606.09859","citing_title":"Mitigating Manifold Departure: Uncertainty-Aware Subspace Rectification for Trustworthy MLLM Decoding","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08245","citing_title":"When Language Overwrites Vision: Over-Alignment and Geometric Debiasing in Vision-Language Models","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08245","citing_title":"When Language Overwrites Vision: Over-Alignment and Geometric Debiasing in Vision-Language Models","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08245","citing_title":"When Language Overwrites Vision: Over-Alignment and Geometric Debiasing in Vision-Language Models","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2404.18930","citing_title":"Hallucination of Multimodal Large Language Models: A Survey","ref_index":74,"is_internal_anchor":false},{"citing_arxiv_id":"2604.10071","citing_title":"Spotlight and Shadow: Attention-Guided Dual-Anchor Introspective Decoding for MLLM Hallucination Mitigation","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2604.21027","citing_title":"HypEHR: Hyperbolic Modeling of Electronic Health Records for Efficient Question Answering","ref_index":290,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/PFXRPNKNVTA5QJCODEM3P7AR3H","json":"https://pith.science/pith/PFXRPNKNVTA5QJCODEM3P7AR3H.json","graph_json":"https://pith.science/api/pith-number/PFXRPNKNVTA5QJCODEM3P7AR3H/graph.json","events_json":"https://pith.science/api/pith-number/PFXRPNKNVTA5QJCODEM3P7AR3H/events.json","paper":"https://pith.science/paper/PFXRPNKN"},"agent_actions":{"view_html":"https://pith.science/pith/PFXRPNKNVTA5QJCODEM3P7AR3H","download_json":"https://pith.science/pith/PFXRPNKNVTA5QJCODEM3P7AR3H.json","view_paper":"https://pith.science/paper/PFXRPNKN","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2408.02032&json=true","fetch_graph":"https://pith.science/api/pith-number/PFXRPNKNVTA5QJCODEM3P7AR3H/graph.json","fetch_events":"https://pith.science/api/pith-number/PFXRPNKNVTA5QJCODEM3P7AR3H/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/PFXRPNKNVTA5QJCODEM3P7AR3H/action/timestamp_anchor","attest_storage":"https://pith.science/pith/PFXRPNKNVTA5QJCODEM3P7AR3H/action/storage_attestation","attest_author":"https://pith.science/pith/PFXRPNKNVTA5QJCODEM3P7AR3H/action/author_attestation","sign_citation":"https://pith.science/pith/PFXRPNKNVTA5QJCODEM3P7AR3H/action/citation_signature","submit_replication":"https://pith.science/pith/PFXRPNKNVTA5QJCODEM3P7AR3H/action/replication_record"}},"created_at":"2026-07-05T10:31:45.648640+00:00","updated_at":"2026-07-05T10:31:45.648640+00:00"}