{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:HO4BZO2CJ3GFLH7R3C7425F7GN","short_pith_number":"pith:HO4BZO2C","schema_version":"1.0","canonical_sha256":"3bb81cbb424ecc559ff1d8bfcd74bf337ad5859c9231b42c1c20cc0509445d62","source":{"kind":"arxiv","id":"2502.01056","version":1},"attestation_state":"computed","paper":{"title":"Mitigating Hallucinations in Large Vision-Language Models with Internal Fact-based Contrastive Decoding","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.CV","authors_text":"Chao Wang, Weiwei Fu, Xuancheng Zhou, Yang Zhou","submitted_at":"2025-02-03T05:08:35Z","abstract_excerpt":"Large Visual Language Models (LVLMs) integrate visual and linguistic modalities, exhibiting exceptional performance across various multimodal tasks. Nevertheless, LVLMs remain vulnerable to the issue of object hallucinations. Previous efforts to mitigate this issue focus on supervised fine-tuning (SFT) or incorporating external knowledge, both of which entail significant costs related to training and the acquisition of external data. To address these challenges, we propose a novel model-agnostic approach termed Internal Fact-based Contrastive Decoding (IFCD), designed to mitigate and suppress "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.01056","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2025-02-03T05:08:35Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"85a535855a235dafe0d4871c0f4bd9543817281db3304503cffb61f045956cf0","abstract_canon_sha256":"de9c8845f1ffd65352182c3733ac0ab4d6d46cf34b66e900e942a966469f30ad"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:08:42.321501Z","signature_b64":"fQCZ5AEee71CCqsG6SUArEwgrxq00IDOqV2yFloSJGDRFWOmVjJ6BDdv8QYni3GTDVgWCkA1V3+7qS3dHUAjCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"3bb81cbb424ecc559ff1d8bfcd74bf337ad5859c9231b42c1c20cc0509445d62","last_reissued_at":"2026-07-05T10:08:42.321073Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:08:42.321073Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Mitigating Hallucinations in Large Vision-Language Models with Internal Fact-based Contrastive Decoding","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.CV","authors_text":"Chao Wang, Weiwei Fu, Xuancheng Zhou, Yang Zhou","submitted_at":"2025-02-03T05:08:35Z","abstract_excerpt":"Large Visual Language Models (LVLMs) integrate visual and linguistic modalities, exhibiting exceptional performance across various multimodal tasks. Nevertheless, LVLMs remain vulnerable to the issue of object hallucinations. Previous efforts to mitigate this issue focus on supervised fine-tuning (SFT) or incorporating external knowledge, both of which entail significant costs related to training and the acquisition of external data. To address these challenges, we propose a novel model-agnostic approach termed Internal Fact-based Contrastive Decoding (IFCD), designed to mitigate and suppress "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.01056","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.01056/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.01056","created_at":"2026-07-05T10:08:42.321130+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.01056v1","created_at":"2026-07-05T10:08:42.321130+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.01056","created_at":"2026-07-05T10:08:42.321130+00:00"},{"alias_kind":"pith_short_12","alias_value":"HO4BZO2CJ3GF","created_at":"2026-07-05T10:08:42.321130+00:00"},{"alias_kind":"pith_short_16","alias_value":"HO4BZO2CJ3GFLH7R","created_at":"2026-07-05T10:08:42.321130+00:00"},{"alias_kind":"pith_short_8","alias_value":"HO4BZO2C","created_at":"2026-07-05T10:08:42.321130+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.01733","citing_title":"GEASS: Gated Evidence-Adaptive Selective Caption Trust for Vision-Language Models","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01733","citing_title":"GEASS: Gated Evidence-Adaptive Selective Caption Trust for Vision-Language Models","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2604.12115","citing_title":"HTDC: Hesitation-Triggered Differential Calibration for Mitigating Hallucination in Large Vision-Language Models","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01733","citing_title":"GEASS: Gated Evidence-Adaptive Selective Caption Trust for Vision-Language Models","ref_index":11,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/HO4BZO2CJ3GFLH7R3C7425F7GN","json":"https://pith.science/pith/HO4BZO2CJ3GFLH7R3C7425F7GN.json","graph_json":"https://pith.science/api/pith-number/HO4BZO2CJ3GFLH7R3C7425F7GN/graph.json","events_json":"https://pith.science/api/pith-number/HO4BZO2CJ3GFLH7R3C7425F7GN/events.json","paper":"https://pith.science/paper/HO4BZO2C"},"agent_actions":{"view_html":"https://pith.science/pith/HO4BZO2CJ3GFLH7R3C7425F7GN","download_json":"https://pith.science/pith/HO4BZO2CJ3GFLH7R3C7425F7GN.json","view_paper":"https://pith.science/paper/HO4BZO2C","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.01056&json=true","fetch_graph":"https://pith.science/api/pith-number/HO4BZO2CJ3GFLH7R3C7425F7GN/graph.json","fetch_events":"https://pith.science/api/pith-number/HO4BZO2CJ3GFLH7R3C7425F7GN/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/HO4BZO2CJ3GFLH7R3C7425F7GN/action/timestamp_anchor","attest_storage":"https://pith.science/pith/HO4BZO2CJ3GFLH7R3C7425F7GN/action/storage_attestation","attest_author":"https://pith.science/pith/HO4BZO2CJ3GFLH7R3C7425F7GN/action/author_attestation","sign_citation":"https://pith.science/pith/HO4BZO2CJ3GFLH7R3C7425F7GN/action/citation_signature","submit_replication":"https://pith.science/pith/HO4BZO2CJ3GFLH7R3C7425F7GN/action/replication_record"}},"created_at":"2026-07-05T10:08:42.321130+00:00","updated_at":"2026-07-05T10:08:42.321130+00:00"}