{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:IO2PAJVPENR5NIQTR3GI7NDXB4","short_pith_number":"pith:IO2PAJVP","schema_version":"1.0","canonical_sha256":"43b4f026af2363d6a2138ecc8fb4770f0845661fd5e43d17b93f08633639ce7d","source":{"kind":"arxiv","id":"2402.14545","version":2},"attestation_state":"computed","paper":{"title":"Less is More: Mitigating Multimodal Hallucination from an EOS Decision Perspective","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CV"],"primary_cat":"cs.CL","authors_text":"Liang Zhang, Qin Jin, Zihao Yue","submitted_at":"2024-02-22T13:33:13Z","abstract_excerpt":"Large Multimodal Models (LMMs) often suffer from multimodal hallucinations, wherein they may create content that is not present in the visual inputs. In this paper, we explore a new angle of this issue: overly detailed training data hinders the model's ability to timely terminate generation, leading to continued outputs beyond visual perception limits. By investigating how the model decides to terminate generation with EOS, the special end-of-sentence token, we find that the model assesses the completeness of the entire sequence by comparing the generated text with the image. This observation "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2402.14545","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-02-22T13:33:13Z","cross_cats_sorted":["cs.CV"],"title_canon_sha256":"0dcbba5fd0796cfd45a4c156c15b3ad110250ebb858f559f4d5e2e4b69e0d74d","abstract_canon_sha256":"c7d240f11d5a46ef86c0720f0cf1808d50f7094603f5d5689a585c331135b5c3"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:24:24.113617Z","signature_b64":"902f9BHXt/IFv7wY+nQN7aRdaplaz8MKR5uVP+YLF8M89ixp9bEXcF7ysxtsvh9A+vhEeNGb5mRHVUaIiLHmBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"43b4f026af2363d6a2138ecc8fb4770f0845661fd5e43d17b93f08633639ce7d","last_reissued_at":"2026-07-05T08:24:24.113149Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:24:24.113149Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Less is More: Mitigating Multimodal Hallucination from an EOS Decision Perspective","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CV"],"primary_cat":"cs.CL","authors_text":"Liang Zhang, Qin Jin, Zihao Yue","submitted_at":"2024-02-22T13:33:13Z","abstract_excerpt":"Large Multimodal Models (LMMs) often suffer from multimodal hallucinations, wherein they may create content that is not present in the visual inputs. In this paper, we explore a new angle of this issue: overly detailed training data hinders the model's ability to timely terminate generation, leading to continued outputs beyond visual perception limits. By investigating how the model decides to terminate generation with EOS, the special end-of-sentence token, we find that the model assesses the completeness of the entire sequence by comparing the generated text with the image. This observation "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2402.14545","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2402.14545/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2402.14545","created_at":"2026-07-05T08:24:24.113204+00:00"},{"alias_kind":"arxiv_version","alias_value":"2402.14545v2","created_at":"2026-07-05T08:24:24.113204+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2402.14545","created_at":"2026-07-05T08:24:24.113204+00:00"},{"alias_kind":"pith_short_12","alias_value":"IO2PAJVPENR5","created_at":"2026-07-05T08:24:24.113204+00:00"},{"alias_kind":"pith_short_16","alias_value":"IO2PAJVPENR5NIQT","created_at":"2026-07-05T08:24:24.113204+00:00"},{"alias_kind":"pith_short_8","alias_value":"IO2PAJVP","created_at":"2026-07-05T08:24:24.113204+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.03376","citing_title":"P$^2$-DPO: Grounding Hallucination in Perceptual Processing via Calibration Direct Preference Optimization","ref_index":114,"is_internal_anchor":false},{"citing_arxiv_id":"2507.12455","citing_title":"Mitigating Object Hallucinations via Sentence-Level Early Intervention","ref_index":79,"is_internal_anchor":false},{"citing_arxiv_id":"2605.21300","citing_title":"Reducing Object Hallucination in LVLMs via Emphasizing Image-negative Tokens","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2605.15300","citing_title":"Deep Pre-Alignment for VLMs","ref_index":82,"is_internal_anchor":false},{"citing_arxiv_id":"2404.18930","citing_title":"Hallucination of Multimodal Large Language Models: A Survey","ref_index":201,"is_internal_anchor":false},{"citing_arxiv_id":"2604.17982","citing_title":"Mitigating Multimodal Hallucination via Phase-wise Self-reward","ref_index":59,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/IO2PAJVPENR5NIQTR3GI7NDXB4","json":"https://pith.science/pith/IO2PAJVPENR5NIQTR3GI7NDXB4.json","graph_json":"https://pith.science/api/pith-number/IO2PAJVPENR5NIQTR3GI7NDXB4/graph.json","events_json":"https://pith.science/api/pith-number/IO2PAJVPENR5NIQTR3GI7NDXB4/events.json","paper":"https://pith.science/paper/IO2PAJVP"},"agent_actions":{"view_html":"https://pith.science/pith/IO2PAJVPENR5NIQTR3GI7NDXB4","download_json":"https://pith.science/pith/IO2PAJVPENR5NIQTR3GI7NDXB4.json","view_paper":"https://pith.science/paper/IO2PAJVP","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2402.14545&json=true","fetch_graph":"https://pith.science/api/pith-number/IO2PAJVPENR5NIQTR3GI7NDXB4/graph.json","fetch_events":"https://pith.science/api/pith-number/IO2PAJVPENR5NIQTR3GI7NDXB4/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/IO2PAJVPENR5NIQTR3GI7NDXB4/action/timestamp_anchor","attest_storage":"https://pith.science/pith/IO2PAJVPENR5NIQTR3GI7NDXB4/action/storage_attestation","attest_author":"https://pith.science/pith/IO2PAJVPENR5NIQTR3GI7NDXB4/action/author_attestation","sign_citation":"https://pith.science/pith/IO2PAJVPENR5NIQTR3GI7NDXB4/action/citation_signature","submit_replication":"https://pith.science/pith/IO2PAJVPENR5NIQTR3GI7NDXB4/action/replication_record"}},"created_at":"2026-07-05T08:24:24.113204+00:00","updated_at":"2026-07-05T08:24:24.113204+00:00"}