{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:ZJ5DO3WSMORTBO7ZYL4AHXZVH6","short_pith_number":"pith:ZJ5DO3WS","schema_version":"1.0","canonical_sha256":"ca7a376ed263a330bbf9c2f803df353f9d46bdc5b307c7898377e8dd234dfcee","source":{"kind":"arxiv","id":"2310.16045","version":2},"attestation_state":"computed","paper":{"title":"Woodpecker: Hallucination Correction for Multimodal Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.LG"],"primary_cat":"cs.CV","authors_text":"Chaoyou Fu, Dianbo Sui, Enhong Chen, Hao Wang, Ke Li, Shukang Yin, Sirui Zhao, Tong Xu, Xing Sun, Yunhang Shen","submitted_at":"2023-10-24T17:58:07Z","abstract_excerpt":"Hallucination is a big shadow hanging over the rapidly evolving Multimodal Large Language Models (MLLMs), referring to the phenomenon that the generated text is inconsistent with the image content. In order to mitigate hallucinations, existing studies mainly resort to an instruction-tuning manner that requires retraining the models with specific data. In this paper, we pave a different way, introducing a training-free method named Woodpecker. Like a woodpecker heals trees, it picks out and corrects hallucinations from the generated text. Concretely, Woodpecker consists of five stages: key conc"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2310.16045","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2023-10-24T17:58:07Z","cross_cats_sorted":["cs.AI","cs.CL","cs.LG"],"title_canon_sha256":"64c2ecb22ea9d34f8553cf615581b2ce0d9c132486ce5e98fbc967985ddc448e","abstract_canon_sha256":"82168156f5d9bc0a458b1aa1e6a7406fcf2938b810fffd8276dfacedbc597e1e"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:47:26.820716Z","signature_b64":"gI6N0ZTHdwYf4ArK/UNwP+PgBpnvLoh1CfG1JWfd9d3+0D6DhS8B+IWXVoFZ46N1rj3VBlG10ttYeU1BrnZZCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ca7a376ed263a330bbf9c2f803df353f9d46bdc5b307c7898377e8dd234dfcee","last_reissued_at":"2026-07-05T09:47:26.820217Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:47:26.820217Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Woodpecker: Hallucination Correction for Multimodal Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.LG"],"primary_cat":"cs.CV","authors_text":"Chaoyou Fu, Dianbo Sui, Enhong Chen, Hao Wang, Ke Li, Shukang Yin, Sirui Zhao, Tong Xu, Xing Sun, Yunhang Shen","submitted_at":"2023-10-24T17:58:07Z","abstract_excerpt":"Hallucination is a big shadow hanging over the rapidly evolving Multimodal Large Language Models (MLLMs), referring to the phenomenon that the generated text is inconsistent with the image content. In order to mitigate hallucinations, existing studies mainly resort to an instruction-tuning manner that requires retraining the models with specific data. In this paper, we pave a different way, introducing a training-free method named Woodpecker. Like a woodpecker heals trees, it picks out and corrects hallucinations from the generated text. Concretely, Woodpecker consists of five stages: key conc"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2310.16045","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2310.16045/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2310.16045","created_at":"2026-07-05T09:47:26.820275+00:00"},{"alias_kind":"arxiv_version","alias_value":"2310.16045v2","created_at":"2026-07-05T09:47:26.820275+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2310.16045","created_at":"2026-07-05T09:47:26.820275+00:00"},{"alias_kind":"pith_short_12","alias_value":"ZJ5DO3WSMORT","created_at":"2026-07-05T09:47:26.820275+00:00"},{"alias_kind":"pith_short_16","alias_value":"ZJ5DO3WSMORTBO7Z","created_at":"2026-07-05T09:47:26.820275+00:00"},{"alias_kind":"pith_short_8","alias_value":"ZJ5DO3WS","created_at":"2026-07-05T09:47:26.820275+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":17,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2410.10238","citing_title":"ForgeryGPT: A Multimodal LLM for Interpretable Image Forgery Detection and Localization","ref_index":61,"is_internal_anchor":false},{"citing_arxiv_id":"2411.16771","citing_title":"VidHal: Benchmarking Temporal Hallucinations in Vision LLMs","ref_index":63,"is_internal_anchor":false},{"citing_arxiv_id":"2605.20033","citing_title":"A Nash Equilibrium Framework For Training-Free Multimodal Step Verification","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2605.15300","citing_title":"Deep Pre-Alignment for VLMs","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2505.21472","citing_title":"Mitigating Hallucination in Large Vision-Language Models via Adaptive Attention Calibration","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2402.11411","citing_title":"Aligning Modalities in Vision Large Language Models via Preference Fine-tuning","ref_index":180,"is_internal_anchor":false},{"citing_arxiv_id":"2401.16420","citing_title":"InternLM-XComposer2: Mastering Free-form Text-Image Composition and Comprehension in Vision-Language Large Model","ref_index":88,"is_internal_anchor":false},{"citing_arxiv_id":"2310.14566","citing_title":"HallusionBench: An Advanced Diagnostic Suite for Entangled Language Hallucination and Visual Illusion in Large Vision-Language Models","ref_index":53,"is_internal_anchor":false},{"citing_arxiv_id":"2311.07397","citing_title":"AMBER: An LLM-free Multi-dimensional Benchmark for MLLMs Hallucination Evaluation","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2306.13549","citing_title":"A Survey on Multimodal Large Language Models","ref_index":79,"is_internal_anchor":false},{"citing_arxiv_id":"2605.13156","citing_title":"Dual-Pathway Circuits of Object Hallucination in Vision-Language Models","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2311.10122","citing_title":"Video-LLaVA: Learning United Visual Representation by Alignment Before Projection","ref_index":120,"is_internal_anchor":false},{"citing_arxiv_id":"2402.00253","citing_title":"A Survey on Hallucination in Large Vision-Language Models","ref_index":47,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10622","citing_title":"Vocabulary Hijacking in LVLMs: Unveiling Critical Attention Heads by Excluding Inert Tokens to Mitigate Hallucination","ref_index":52,"is_internal_anchor":false},{"citing_arxiv_id":"2404.18930","citing_title":"Hallucination of Multimodal Large Language Models: A Survey","ref_index":192,"is_internal_anchor":false},{"citing_arxiv_id":"2306.13394","citing_title":"MME: A Comprehensive Evaluation Benchmark for Multimodal Large Language Models","ref_index":54,"is_internal_anchor":false},{"citing_arxiv_id":"2604.20366","citing_title":"Mitigating Hallucinations in Large Vision-Language Models without Performance Degradation","ref_index":144,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ZJ5DO3WSMORTBO7ZYL4AHXZVH6","json":"https://pith.science/pith/ZJ5DO3WSMORTBO7ZYL4AHXZVH6.json","graph_json":"https://pith.science/api/pith-number/ZJ5DO3WSMORTBO7ZYL4AHXZVH6/graph.json","events_json":"https://pith.science/api/pith-number/ZJ5DO3WSMORTBO7ZYL4AHXZVH6/events.json","paper":"https://pith.science/paper/ZJ5DO3WS"},"agent_actions":{"view_html":"https://pith.science/pith/ZJ5DO3WSMORTBO7ZYL4AHXZVH6","download_json":"https://pith.science/pith/ZJ5DO3WSMORTBO7ZYL4AHXZVH6.json","view_paper":"https://pith.science/paper/ZJ5DO3WS","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2310.16045&json=true","fetch_graph":"https://pith.science/api/pith-number/ZJ5DO3WSMORTBO7ZYL4AHXZVH6/graph.json","fetch_events":"https://pith.science/api/pith-number/ZJ5DO3WSMORTBO7ZYL4AHXZVH6/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ZJ5DO3WSMORTBO7ZYL4AHXZVH6/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ZJ5DO3WSMORTBO7ZYL4AHXZVH6/action/storage_attestation","attest_author":"https://pith.science/pith/ZJ5DO3WSMORTBO7ZYL4AHXZVH6/action/author_attestation","sign_citation":"https://pith.science/pith/ZJ5DO3WSMORTBO7ZYL4AHXZVH6/action/citation_signature","submit_replication":"https://pith.science/pith/ZJ5DO3WSMORTBO7ZYL4AHXZVH6/action/replication_record"}},"created_at":"2026-07-05T09:47:26.820275+00:00","updated_at":"2026-07-05T09:47:26.820275+00:00"}