{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:U6EDSYK3M5P3QWF4YUEGU3QMZ5","short_pith_number":"pith:U6EDSYK3","schema_version":"1.0","canonical_sha256":"a78839615b675fb858bcc5086a6e0ccf7059f10e99d8f355825e32bc520ebc16","source":{"kind":"arxiv","id":"2503.06486","version":1},"attestation_state":"computed","paper":{"title":"PerturboLLaVA: Reducing Multimodal Hallucinations with Perturbative Visual Training","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Bo Zhang, Chenchen Jing, Chunhua Shen, Cong Chen, Fengyun Rao, Hao Chen, Mingyu Liu, Yizhou Zhou","submitted_at":"2025-03-09T07:07:03Z","abstract_excerpt":"This paper aims to address the challenge of hallucinations in Multimodal Large Language Models (MLLMs) particularly for dense image captioning tasks. To tackle the challenge, we identify the current lack of a metric that finely measures the caption quality in concept level. We hereby introduce HalFscore, a novel metric built upon the language graph and is designed to evaluate both the accuracy and completeness of dense captions at a granular level. Additionally, we identify the root cause of hallucination as the model's over-reliance on its language prior. To address this, we propose PerturboL"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2503.06486","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2025-03-09T07:07:03Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"4407fee008cbdb0ea314bfdef3767463c8de1c5ec367263381356634f5ee0b91","abstract_canon_sha256":"04540f22d8d0e341dfee01da002d7d2c279e8804a1b1df07d00f1c5c4e2c1da2"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:27:35.079824Z","signature_b64":"HELCyf8rDaWgllxmUxryrNNxMMVl8l/3h0iiD1g4KGjR8TFcr5A3kbWhvPBiHoWN0TtS3w58EyuXva+P3PScBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"a78839615b675fb858bcc5086a6e0ccf7059f10e99d8f355825e32bc520ebc16","last_reissued_at":"2026-07-05T10:27:35.079188Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:27:35.079188Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"PerturboLLaVA: Reducing Multimodal Hallucinations with Perturbative Visual Training","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Bo Zhang, Chenchen Jing, Chunhua Shen, Cong Chen, Fengyun Rao, Hao Chen, Mingyu Liu, Yizhou Zhou","submitted_at":"2025-03-09T07:07:03Z","abstract_excerpt":"This paper aims to address the challenge of hallucinations in Multimodal Large Language Models (MLLMs) particularly for dense image captioning tasks. To tackle the challenge, we identify the current lack of a metric that finely measures the caption quality in concept level. We hereby introduce HalFscore, a novel metric built upon the language graph and is designed to evaluate both the accuracy and completeness of dense captions at a granular level. Additionally, we identify the root cause of hallucination as the model's over-reliance on its language prior. To address this, we propose PerturboL"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2503.06486","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2503.06486/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2503.06486","created_at":"2026-07-05T10:27:35.079257+00:00"},{"alias_kind":"arxiv_version","alias_value":"2503.06486v1","created_at":"2026-07-05T10:27:35.079257+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2503.06486","created_at":"2026-07-05T10:27:35.079257+00:00"},{"alias_kind":"pith_short_12","alias_value":"U6EDSYK3M5P3","created_at":"2026-07-05T10:27:35.079257+00:00"},{"alias_kind":"pith_short_16","alias_value":"U6EDSYK3M5P3QWF4","created_at":"2026-07-05T10:27:35.079257+00:00"},{"alias_kind":"pith_short_8","alias_value":"U6EDSYK3","created_at":"2026-07-05T10:27:35.079257+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2602.17555","citing_title":"GraphThinker: Reinforcing Temporally Grounded Video Reasoning with Event Graph Thinking","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2404.18930","citing_title":"Hallucination of Multimodal Large Language Models: A Survey","ref_index":24,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/U6EDSYK3M5P3QWF4YUEGU3QMZ5","json":"https://pith.science/pith/U6EDSYK3M5P3QWF4YUEGU3QMZ5.json","graph_json":"https://pith.science/api/pith-number/U6EDSYK3M5P3QWF4YUEGU3QMZ5/graph.json","events_json":"https://pith.science/api/pith-number/U6EDSYK3M5P3QWF4YUEGU3QMZ5/events.json","paper":"https://pith.science/paper/U6EDSYK3"},"agent_actions":{"view_html":"https://pith.science/pith/U6EDSYK3M5P3QWF4YUEGU3QMZ5","download_json":"https://pith.science/pith/U6EDSYK3M5P3QWF4YUEGU3QMZ5.json","view_paper":"https://pith.science/paper/U6EDSYK3","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2503.06486&json=true","fetch_graph":"https://pith.science/api/pith-number/U6EDSYK3M5P3QWF4YUEGU3QMZ5/graph.json","fetch_events":"https://pith.science/api/pith-number/U6EDSYK3M5P3QWF4YUEGU3QMZ5/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/U6EDSYK3M5P3QWF4YUEGU3QMZ5/action/timestamp_anchor","attest_storage":"https://pith.science/pith/U6EDSYK3M5P3QWF4YUEGU3QMZ5/action/storage_attestation","attest_author":"https://pith.science/pith/U6EDSYK3M5P3QWF4YUEGU3QMZ5/action/author_attestation","sign_citation":"https://pith.science/pith/U6EDSYK3M5P3QWF4YUEGU3QMZ5/action/citation_signature","submit_replication":"https://pith.science/pith/U6EDSYK3M5P3QWF4YUEGU3QMZ5/action/replication_record"}},"created_at":"2026-07-05T10:27:35.079257+00:00","updated_at":"2026-07-05T10:27:35.079257+00:00"}