{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:TL4H77GRG3TWTGOOSX32GILN27","short_pith_number":"pith:TL4H77GR","schema_version":"1.0","canonical_sha256":"9af87ffcd136e76999ce95f7a3216dd7e58529e27d4e997a2dd06ebbc5774722","source":{"kind":"arxiv","id":"2406.14492","version":1},"attestation_state":"computed","paper":{"title":"Does Object Grounding Really Reduce Hallucination of Large Vision-Language Models?","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.CV","authors_text":"Goran Glava\\v{s}, Gregor Geigle, Radu Timofte","submitted_at":"2024-06-20T16:56:11Z","abstract_excerpt":"Large vision-language models (LVLMs) have recently dramatically pushed the state of the art in image captioning and many image understanding tasks (e.g., visual question answering). LVLMs, however, often \\textit{hallucinate} and produce captions that mention concepts that cannot be found in the image. These hallucinations erode the trustworthiness of LVLMs and are arguably among the main obstacles to their ubiquitous adoption. Recent work suggests that addition of grounding objectives -- those that explicitly align image regions or objects to text spans -- reduces the amount of LVLM hallucinat"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2406.14492","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-06-20T16:56:11Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"b504993cb37173bef9bf7450ec19aa4e85617a8dc7219d0948dff49549ef9e00","abstract_canon_sha256":"1215fb8b8fa722eb366d1c58d5991cd6e4734c4801571f00ad90b89b2598e10d"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:34:51.501602Z","signature_b64":"h84HHbCzipBp4rW/wS1/4MSOMVZtRNb+8Roj4sJbGTDPwNZFc1dd2e0H6ZkqvgNhXEH71qA3jvOioLZRw+Y1Bg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"9af87ffcd136e76999ce95f7a3216dd7e58529e27d4e997a2dd06ebbc5774722","last_reissued_at":"2026-07-05T08:34:51.501155Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:34:51.501155Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Does Object Grounding Really Reduce Hallucination of Large Vision-Language Models?","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.CV","authors_text":"Goran Glava\\v{s}, Gregor Geigle, Radu Timofte","submitted_at":"2024-06-20T16:56:11Z","abstract_excerpt":"Large vision-language models (LVLMs) have recently dramatically pushed the state of the art in image captioning and many image understanding tasks (e.g., visual question answering). LVLMs, however, often \\textit{hallucinate} and produce captions that mention concepts that cannot be found in the image. These hallucinations erode the trustworthiness of LVLMs and are arguably among the main obstacles to their ubiquitous adoption. Recent work suggests that addition of grounding objectives -- those that explicitly align image regions or objects to text spans -- reduces the amount of LVLM hallucinat"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.14492","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.14492/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2406.14492","created_at":"2026-07-05T08:34:51.501209+00:00"},{"alias_kind":"arxiv_version","alias_value":"2406.14492v1","created_at":"2026-07-05T08:34:51.501209+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.14492","created_at":"2026-07-05T08:34:51.501209+00:00"},{"alias_kind":"pith_short_12","alias_value":"TL4H77GRG3TW","created_at":"2026-07-05T08:34:51.501209+00:00"},{"alias_kind":"pith_short_16","alias_value":"TL4H77GRG3TWTGOO","created_at":"2026-07-05T08:34:51.501209+00:00"},{"alias_kind":"pith_short_8","alias_value":"TL4H77GR","created_at":"2026-07-05T08:34:51.501209+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/TL4H77GRG3TWTGOOSX32GILN27","json":"https://pith.science/pith/TL4H77GRG3TWTGOOSX32GILN27.json","graph_json":"https://pith.science/api/pith-number/TL4H77GRG3TWTGOOSX32GILN27/graph.json","events_json":"https://pith.science/api/pith-number/TL4H77GRG3TWTGOOSX32GILN27/events.json","paper":"https://pith.science/paper/TL4H77GR"},"agent_actions":{"view_html":"https://pith.science/pith/TL4H77GRG3TWTGOOSX32GILN27","download_json":"https://pith.science/pith/TL4H77GRG3TWTGOOSX32GILN27.json","view_paper":"https://pith.science/paper/TL4H77GR","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2406.14492&json=true","fetch_graph":"https://pith.science/api/pith-number/TL4H77GRG3TWTGOOSX32GILN27/graph.json","fetch_events":"https://pith.science/api/pith-number/TL4H77GRG3TWTGOOSX32GILN27/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/TL4H77GRG3TWTGOOSX32GILN27/action/timestamp_anchor","attest_storage":"https://pith.science/pith/TL4H77GRG3TWTGOOSX32GILN27/action/storage_attestation","attest_author":"https://pith.science/pith/TL4H77GRG3TWTGOOSX32GILN27/action/author_attestation","sign_citation":"https://pith.science/pith/TL4H77GRG3TWTGOOSX32GILN27/action/citation_signature","submit_replication":"https://pith.science/pith/TL4H77GRG3TWTGOOSX32GILN27/action/replication_record"}},"created_at":"2026-07-05T08:34:51.501209+00:00","updated_at":"2026-07-05T08:34:51.501209+00:00"}