{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:LZ7FW3M27GGOX5EIXVXNAQUUH5","short_pith_number":"pith:LZ7FW3M2","schema_version":"1.0","canonical_sha256":"5e7e5b6d9af98cebf488bd6ed042943f6a356df4ee8b370b02929ecd250aa36d","source":{"kind":"arxiv","id":"2403.14401","version":2},"attestation_state":"computed","paper":{"title":"Pensieve: Retrospect-then-Compare Mitigates Visual Hallucination","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Bowen Cao, Changjun Jiang, Dingchen Yang, Guang Chen","submitted_at":"2024-03-21T13:49:42Z","abstract_excerpt":"Multi-modal Large Language Models (MLLMs) demonstrate remarkable success across various vision-language tasks. However, they suffer from visual hallucination, where the generated responses diverge from the provided image. Are MLLMs oblivious to the accurate visual cues when they hallucinate? Our investigation reveals that the visual branch may equally advocate both accurate and erroneous content. To address this issue, we propose Pensieve, a training-free method that leverages the analogous visual hallucinations, which are induced by images sharing common semantic and appearance characteristic"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2403.14401","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-03-21T13:49:42Z","cross_cats_sorted":[],"title_canon_sha256":"0ac846589d4075eda5088f405c996fd46905ab2798413ffa7f3bd8a47ff68067","abstract_canon_sha256":"7270472b9bf45d0072c8bc8e83108db09363c50e168533a92a0795e91852caca"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:01:31.410675Z","signature_b64":"+errQDj4M7CoNii1AdyrSOTWDn48Md/82xa+l1VTwvWWbH/1LypItZ/mBVEwAg/QCokR8KQeo4r89jPSffDsCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"5e7e5b6d9af98cebf488bd6ed042943f6a356df4ee8b370b02929ecd250aa36d","last_reissued_at":"2026-07-05T09:01:31.410199Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:01:31.410199Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Pensieve: Retrospect-then-Compare Mitigates Visual Hallucination","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Bowen Cao, Changjun Jiang, Dingchen Yang, Guang Chen","submitted_at":"2024-03-21T13:49:42Z","abstract_excerpt":"Multi-modal Large Language Models (MLLMs) demonstrate remarkable success across various vision-language tasks. However, they suffer from visual hallucination, where the generated responses diverge from the provided image. Are MLLMs oblivious to the accurate visual cues when they hallucinate? Our investigation reveals that the visual branch may equally advocate both accurate and erroneous content. To address this issue, we propose Pensieve, a training-free method that leverages the analogous visual hallucinations, which are induced by images sharing common semantic and appearance characteristic"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2403.14401","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2403.14401/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2403.14401","created_at":"2026-07-05T09:01:31.410253+00:00"},{"alias_kind":"arxiv_version","alias_value":"2403.14401v2","created_at":"2026-07-05T09:01:31.410253+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2403.14401","created_at":"2026-07-05T09:01:31.410253+00:00"},{"alias_kind":"pith_short_12","alias_value":"LZ7FW3M27GGO","created_at":"2026-07-05T09:01:31.410253+00:00"},{"alias_kind":"pith_short_16","alias_value":"LZ7FW3M27GGOX5EI","created_at":"2026-07-05T09:01:31.410253+00:00"},{"alias_kind":"pith_short_8","alias_value":"LZ7FW3M2","created_at":"2026-07-05T09:01:31.410253+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.31429","citing_title":"YARD: Y-Architecture Register Decoding for Efficient Hallucination Mitigation in Large Vision-Language Models","ref_index":49,"is_internal_anchor":false},{"citing_arxiv_id":"2604.21027","citing_title":"HypEHR: Hyperbolic Modeling of Electronic Health Records for Efficient Question Answering","ref_index":268,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/LZ7FW3M27GGOX5EIXVXNAQUUH5","json":"https://pith.science/pith/LZ7FW3M27GGOX5EIXVXNAQUUH5.json","graph_json":"https://pith.science/api/pith-number/LZ7FW3M27GGOX5EIXVXNAQUUH5/graph.json","events_json":"https://pith.science/api/pith-number/LZ7FW3M27GGOX5EIXVXNAQUUH5/events.json","paper":"https://pith.science/paper/LZ7FW3M2"},"agent_actions":{"view_html":"https://pith.science/pith/LZ7FW3M27GGOX5EIXVXNAQUUH5","download_json":"https://pith.science/pith/LZ7FW3M27GGOX5EIXVXNAQUUH5.json","view_paper":"https://pith.science/paper/LZ7FW3M2","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2403.14401&json=true","fetch_graph":"https://pith.science/api/pith-number/LZ7FW3M27GGOX5EIXVXNAQUUH5/graph.json","fetch_events":"https://pith.science/api/pith-number/LZ7FW3M27GGOX5EIXVXNAQUUH5/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/LZ7FW3M27GGOX5EIXVXNAQUUH5/action/timestamp_anchor","attest_storage":"https://pith.science/pith/LZ7FW3M27GGOX5EIXVXNAQUUH5/action/storage_attestation","attest_author":"https://pith.science/pith/LZ7FW3M27GGOX5EIXVXNAQUUH5/action/author_attestation","sign_citation":"https://pith.science/pith/LZ7FW3M27GGOX5EIXVXNAQUUH5/action/citation_signature","submit_replication":"https://pith.science/pith/LZ7FW3M27GGOX5EIXVXNAQUUH5/action/replication_record"}},"created_at":"2026-07-05T09:01:31.410253+00:00","updated_at":"2026-07-05T09:01:31.410253+00:00"}