{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:JEDU7KU3IORW6QTOYJD4ZK3ULW","short_pith_number":"pith:JEDU7KU3","schema_version":"1.0","canonical_sha256":"49074faa9b43a36f426ec247ccab745db42c747e9e8f5ea8659a508f10d79c81","source":{"kind":"arxiv","id":"2405.15683","version":3},"attestation_state":"computed","paper":{"title":"Visual Description Grounding Reduces Hallucinations and Boosts Reasoning in LVLMs","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.CV","authors_text":"Chandra Kiran Reddy Evuru, Dinesh Manocha, Oriol Nieto, Sonal Kumar, Sreyan Ghosh, Utkarsh Tyagi, Zeyu Jin","submitted_at":"2024-05-24T16:21:59Z","abstract_excerpt":"Large Vision-Language Models (LVLMs) often produce responses that misalign with factual information, a phenomenon known as hallucinations. While hallucinations are well-studied, the exact causes behind them remain underexplored. In this paper, we first investigate the root causes of hallucinations in LVLMs. Our findings reveal that existing mitigation techniques primarily reduce hallucinations for visual recognition prompts-those that require simple descriptions of visual elements-but fail for cognitive prompts that demand deliberate reasoning. We identify the core issue as a lack of true visu"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2405.15683","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2024-05-24T16:21:59Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"196b3d9b36e70f119e77d6069f73797ea92972fccb268a684ddfde61c87bdbb3","abstract_canon_sha256":"f695d6c67b53e0d52cf2a7525395dd1655447d30707575763f3019ed424836fe"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:25:06.750280Z","signature_b64":"WR4Zf7pPYv7W4X5L5U0YvHdRk4Gg71k2RZDOr+rF28s5LnGWEdqNC7dIXSDA/zXVpNf/9RORa/FU3C0X0ypDDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"49074faa9b43a36f426ec247ccab745db42c747e9e8f5ea8659a508f10d79c81","last_reissued_at":"2026-07-05T10:25:06.749652Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:25:06.749652Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Visual Description Grounding Reduces Hallucinations and Boosts Reasoning in LVLMs","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.CV","authors_text":"Chandra Kiran Reddy Evuru, Dinesh Manocha, Oriol Nieto, Sonal Kumar, Sreyan Ghosh, Utkarsh Tyagi, Zeyu Jin","submitted_at":"2024-05-24T16:21:59Z","abstract_excerpt":"Large Vision-Language Models (LVLMs) often produce responses that misalign with factual information, a phenomenon known as hallucinations. While hallucinations are well-studied, the exact causes behind them remain underexplored. In this paper, we first investigate the root causes of hallucinations in LVLMs. Our findings reveal that existing mitigation techniques primarily reduce hallucinations for visual recognition prompts-those that require simple descriptions of visual elements-but fail for cognitive prompts that demand deliberate reasoning. We identify the core issue as a lack of true visu"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2405.15683","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2405.15683/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2405.15683","created_at":"2026-07-05T10:25:06.749722+00:00"},{"alias_kind":"arxiv_version","alias_value":"2405.15683v3","created_at":"2026-07-05T10:25:06.749722+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2405.15683","created_at":"2026-07-05T10:25:06.749722+00:00"},{"alias_kind":"pith_short_12","alias_value":"JEDU7KU3IORW","created_at":"2026-07-05T10:25:06.749722+00:00"},{"alias_kind":"pith_short_16","alias_value":"JEDU7KU3IORW6QTO","created_at":"2026-07-05T10:25:06.749722+00:00"},{"alias_kind":"pith_short_8","alias_value":"JEDU7KU3","created_at":"2026-07-05T10:25:06.749722+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2604.23443","citing_title":"Revisiting Greedy Decoding for Visual Question Answering: A Calibration Perspective","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2606.19927","citing_title":"CARE: Competence-Aware Reward Shaping for Adaptive Reasoning Length in Video-MLLMs","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2605.21924","citing_title":"Visual-Advantage On-Policy Distillation for Vision-Language Models","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2604.23443","citing_title":"Revisiting Greedy Decoding for Visual Question Answering: A Calibration Perspective","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2604.21027","citing_title":"HypEHR: Hyperbolic Modeling of Electronic Health Records for Efficient Question Answering","ref_index":281,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/JEDU7KU3IORW6QTOYJD4ZK3ULW","json":"https://pith.science/pith/JEDU7KU3IORW6QTOYJD4ZK3ULW.json","graph_json":"https://pith.science/api/pith-number/JEDU7KU3IORW6QTOYJD4ZK3ULW/graph.json","events_json":"https://pith.science/api/pith-number/JEDU7KU3IORW6QTOYJD4ZK3ULW/events.json","paper":"https://pith.science/paper/JEDU7KU3"},"agent_actions":{"view_html":"https://pith.science/pith/JEDU7KU3IORW6QTOYJD4ZK3ULW","download_json":"https://pith.science/pith/JEDU7KU3IORW6QTOYJD4ZK3ULW.json","view_paper":"https://pith.science/paper/JEDU7KU3","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2405.15683&json=true","fetch_graph":"https://pith.science/api/pith-number/JEDU7KU3IORW6QTOYJD4ZK3ULW/graph.json","fetch_events":"https://pith.science/api/pith-number/JEDU7KU3IORW6QTOYJD4ZK3ULW/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/JEDU7KU3IORW6QTOYJD4ZK3ULW/action/timestamp_anchor","attest_storage":"https://pith.science/pith/JEDU7KU3IORW6QTOYJD4ZK3ULW/action/storage_attestation","attest_author":"https://pith.science/pith/JEDU7KU3IORW6QTOYJD4ZK3ULW/action/author_attestation","sign_citation":"https://pith.science/pith/JEDU7KU3IORW6QTOYJD4ZK3ULW/action/citation_signature","submit_replication":"https://pith.science/pith/JEDU7KU3IORW6QTOYJD4ZK3ULW/action/replication_record"}},"created_at":"2026-07-05T10:25:06.749722+00:00","updated_at":"2026-07-05T10:25:06.749722+00:00"}