{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:3BZRS5BHITILO7TDQM636F5VG5","short_pith_number":"pith:3BZRS5BH","schema_version":"1.0","canonical_sha256":"d87319742744d0b77e63833dbf17b5377f54a501b12160b28a1e3b6bb81f19c0","source":{"kind":"arxiv","id":"2409.20429","version":1},"attestation_state":"computed","paper":{"title":"HELPD: Mitigating Hallucination of LVLMs by Hierarchical Feedback Learning with Vision-enhanced Penalty Decoding","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CV"],"primary_cat":"cs.CL","authors_text":"Chi Qin, Fan Yuan, Piji Li, Xiaogang Xu","submitted_at":"2024-09-30T15:52:05Z","abstract_excerpt":"Large Vision-Language Models (LVLMs) have shown remarkable performance on many visual-language tasks. However, these models still suffer from multimodal hallucination, which means the generation of objects or content that violates the images. Many existing work detects hallucination by directly judging whether an object exists in an image, overlooking the association between the object and semantics. To address this issue, we propose Hierarchical Feedback Learning with Vision-enhanced Penalty Decoding (HELPD). This framework incorporates hallucination feedback at both object and sentence seman"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2409.20429","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-09-30T15:52:05Z","cross_cats_sorted":["cs.CV"],"title_canon_sha256":"ef59395d6b60eaf69078e5d8c94252447f0b33dc0f1700fe3e705e68749a2ca6","abstract_canon_sha256":"b008a1dcae59c2e87bd03d4b15807abbd71c4b2155b2c79f6164739e7318d00e"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:13:43.425895Z","signature_b64":"h3L0lceRx7NhGaQ85zIFAHpwQCcUPVZq1kgLnlJ57jdhyzEvidgrD3aOhxBQTY76dtt+jP1LLGmu7G1z6coDDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"d87319742744d0b77e63833dbf17b5377f54a501b12160b28a1e3b6bb81f19c0","last_reissued_at":"2026-07-05T09:13:43.425431Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:13:43.425431Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"HELPD: Mitigating Hallucination of LVLMs by Hierarchical Feedback Learning with Vision-enhanced Penalty Decoding","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CV"],"primary_cat":"cs.CL","authors_text":"Chi Qin, Fan Yuan, Piji Li, Xiaogang Xu","submitted_at":"2024-09-30T15:52:05Z","abstract_excerpt":"Large Vision-Language Models (LVLMs) have shown remarkable performance on many visual-language tasks. However, these models still suffer from multimodal hallucination, which means the generation of objects or content that violates the images. Many existing work detects hallucination by directly judging whether an object exists in an image, overlooking the association between the object and semantics. To address this issue, we propose Hierarchical Feedback Learning with Vision-enhanced Penalty Decoding (HELPD). This framework incorporates hallucination feedback at both object and sentence seman"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2409.20429","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2409.20429/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2409.20429","created_at":"2026-07-05T09:13:43.425485+00:00"},{"alias_kind":"arxiv_version","alias_value":"2409.20429v1","created_at":"2026-07-05T09:13:43.425485+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2409.20429","created_at":"2026-07-05T09:13:43.425485+00:00"},{"alias_kind":"pith_short_12","alias_value":"3BZRS5BHITIL","created_at":"2026-07-05T09:13:43.425485+00:00"},{"alias_kind":"pith_short_16","alias_value":"3BZRS5BHITILO7TD","created_at":"2026-07-05T09:13:43.425485+00:00"},{"alias_kind":"pith_short_8","alias_value":"3BZRS5BH","created_at":"2026-07-05T09:13:43.425485+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2404.18930","citing_title":"Hallucination of Multimodal Large Language Models: A Survey","ref_index":200,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/3BZRS5BHITILO7TDQM636F5VG5","json":"https://pith.science/pith/3BZRS5BHITILO7TDQM636F5VG5.json","graph_json":"https://pith.science/api/pith-number/3BZRS5BHITILO7TDQM636F5VG5/graph.json","events_json":"https://pith.science/api/pith-number/3BZRS5BHITILO7TDQM636F5VG5/events.json","paper":"https://pith.science/paper/3BZRS5BH"},"agent_actions":{"view_html":"https://pith.science/pith/3BZRS5BHITILO7TDQM636F5VG5","download_json":"https://pith.science/pith/3BZRS5BHITILO7TDQM636F5VG5.json","view_paper":"https://pith.science/paper/3BZRS5BH","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2409.20429&json=true","fetch_graph":"https://pith.science/api/pith-number/3BZRS5BHITILO7TDQM636F5VG5/graph.json","fetch_events":"https://pith.science/api/pith-number/3BZRS5BHITILO7TDQM636F5VG5/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/3BZRS5BHITILO7TDQM636F5VG5/action/timestamp_anchor","attest_storage":"https://pith.science/pith/3BZRS5BHITILO7TDQM636F5VG5/action/storage_attestation","attest_author":"https://pith.science/pith/3BZRS5BHITILO7TDQM636F5VG5/action/author_attestation","sign_citation":"https://pith.science/pith/3BZRS5BHITILO7TDQM636F5VG5/action/citation_signature","submit_replication":"https://pith.science/pith/3BZRS5BHITILO7TDQM636F5VG5/action/replication_record"}},"created_at":"2026-07-05T09:13:43.425485+00:00","updated_at":"2026-07-05T09:13:43.425485+00:00"}