{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:MKK64FZNUHAIXP2HKHUY5IL4AA","short_pith_number":"pith:MKK64FZN","schema_version":"1.0","canonical_sha256":"6295ee172da1c08bbf4751e98ea17c0016c228aab6502d9753dd37a60e898f6d","source":{"kind":"arxiv","id":"2312.00849","version":2},"attestation_state":"computed","paper":{"title":"RLHF-V: Towards Trustworthy MLLMs via Behavior Alignment from Fine-grained Correctional Human Feedback","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CV"],"primary_cat":"cs.CL","authors_text":"Ganqu Cui, Hai-Tao Zheng, Haoye Zhang, Jinyi Hu, Maosong Sun, Taiwen He, Tat-Seng Chua, Tianyu Yu, Yifeng Han, Yuan Yao, Zhiyuan Liu","submitted_at":"2023-12-01T11:36:08Z","abstract_excerpt":"Multimodal Large Language Models (MLLMs) have recently demonstrated impressive capabilities in multimodal understanding, reasoning, and interaction. However, existing MLLMs prevalently suffer from serious hallucination problems, generating text that is not factually grounded in associated images. The problem makes existing MLLMs untrustworthy and thus impractical in real-world (especially high-stakes) applications. To address the challenge, we present RLHF-V, which enhances MLLM trustworthiness via behavior alignment from fine-grained correctional human feedback. Specifically, RLHF-V collects "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2312.00849","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2023-12-01T11:36:08Z","cross_cats_sorted":["cs.CV"],"title_canon_sha256":"9f7008b703560b6d1a8470ff8870be2cd83def1d1f0e93543fd2176eaa583ac1","abstract_canon_sha256":"5df022035fb9fe5572afaf1be2a2f5b1aa2c84bd0d5a6ece958d0e92ec48a310"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:53:32.250185Z","signature_b64":"PwTA/MIcFzcoHetuQmfyHs+ec3I9qnet7NEjm1/v6lu7LRt64oCXW/Z7ZevEtg2hoHNS2vCdjxoLZO074kbaCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"6295ee172da1c08bbf4751e98ea17c0016c228aab6502d9753dd37a60e898f6d","last_reissued_at":"2026-07-05T07:53:32.249759Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:53:32.249759Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"RLHF-V: Towards Trustworthy MLLMs via Behavior Alignment from Fine-grained Correctional Human Feedback","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CV"],"primary_cat":"cs.CL","authors_text":"Ganqu Cui, Hai-Tao Zheng, Haoye Zhang, Jinyi Hu, Maosong Sun, Taiwen He, Tat-Seng Chua, Tianyu Yu, Yifeng Han, Yuan Yao, Zhiyuan Liu","submitted_at":"2023-12-01T11:36:08Z","abstract_excerpt":"Multimodal Large Language Models (MLLMs) have recently demonstrated impressive capabilities in multimodal understanding, reasoning, and interaction. However, existing MLLMs prevalently suffer from serious hallucination problems, generating text that is not factually grounded in associated images. The problem makes existing MLLMs untrustworthy and thus impractical in real-world (especially high-stakes) applications. To address the challenge, we present RLHF-V, which enhances MLLM trustworthiness via behavior alignment from fine-grained correctional human feedback. Specifically, RLHF-V collects "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2312.00849","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2312.00849/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2312.00849","created_at":"2026-07-05T07:53:32.249815+00:00"},{"alias_kind":"arxiv_version","alias_value":"2312.00849v2","created_at":"2026-07-05T07:53:32.249815+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2312.00849","created_at":"2026-07-05T07:53:32.249815+00:00"},{"alias_kind":"pith_short_12","alias_value":"MKK64FZNUHAI","created_at":"2026-07-05T07:53:32.249815+00:00"},{"alias_kind":"pith_short_16","alias_value":"MKK64FZNUHAIXP2H","created_at":"2026-07-05T07:53:32.249815+00:00"},{"alias_kind":"pith_short_8","alias_value":"MKK64FZN","created_at":"2026-07-05T07:53:32.249815+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":10,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.25343","citing_title":"Toward Native Multimodal Modeling: A Roadmap","ref_index":140,"is_internal_anchor":false},{"citing_arxiv_id":"2605.27993","citing_title":"Rethinking Visual Neglect: Steering via Context-Preference for MLLM Hallucination Mitigation","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2402.11411","citing_title":"Aligning Modalities in Vision Large Language Models via Preference Fine-tuning","ref_index":182,"is_internal_anchor":false},{"citing_arxiv_id":"2412.14164","citing_title":"MetaMorph: Multimodal Understanding and Generation via Instruction Tuning","ref_index":159,"is_internal_anchor":false},{"citing_arxiv_id":"2406.16860","citing_title":"Cambrian-1: A Fully Open, Vision-Centric Exploration of Multimodal LLMs","ref_index":141,"is_internal_anchor":false},{"citing_arxiv_id":"2306.13549","citing_title":"A Survey on Multimodal Large Language Models","ref_index":116,"is_internal_anchor":false},{"citing_arxiv_id":"2605.13156","citing_title":"Dual-Pathway Circuits of Object Hallucination in Vision-Language Models","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2402.00253","citing_title":"A Survey on Hallucination in Large Vision-Language Models","ref_index":49,"is_internal_anchor":false},{"citing_arxiv_id":"2404.18930","citing_title":"Hallucination of Multimodal Large Language Models: A Survey","ref_index":198,"is_internal_anchor":false},{"citing_arxiv_id":"2605.04641","citing_title":"CAST: Mitigating Object Hallucination in Large Vision-Language Models via Caption-Guided Visual Attention Steering","ref_index":89,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/MKK64FZNUHAIXP2HKHUY5IL4AA","json":"https://pith.science/pith/MKK64FZNUHAIXP2HKHUY5IL4AA.json","graph_json":"https://pith.science/api/pith-number/MKK64FZNUHAIXP2HKHUY5IL4AA/graph.json","events_json":"https://pith.science/api/pith-number/MKK64FZNUHAIXP2HKHUY5IL4AA/events.json","paper":"https://pith.science/paper/MKK64FZN"},"agent_actions":{"view_html":"https://pith.science/pith/MKK64FZNUHAIXP2HKHUY5IL4AA","download_json":"https://pith.science/pith/MKK64FZNUHAIXP2HKHUY5IL4AA.json","view_paper":"https://pith.science/paper/MKK64FZN","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2312.00849&json=true","fetch_graph":"https://pith.science/api/pith-number/MKK64FZNUHAIXP2HKHUY5IL4AA/graph.json","fetch_events":"https://pith.science/api/pith-number/MKK64FZNUHAIXP2HKHUY5IL4AA/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/MKK64FZNUHAIXP2HKHUY5IL4AA/action/timestamp_anchor","attest_storage":"https://pith.science/pith/MKK64FZNUHAIXP2HKHUY5IL4AA/action/storage_attestation","attest_author":"https://pith.science/pith/MKK64FZNUHAIXP2HKHUY5IL4AA/action/author_attestation","sign_citation":"https://pith.science/pith/MKK64FZNUHAIXP2HKHUY5IL4AA/action/citation_signature","submit_replication":"https://pith.science/pith/MKK64FZNUHAIXP2HKHUY5IL4AA/action/replication_record"}},"created_at":"2026-07-05T07:53:32.249815+00:00","updated_at":"2026-07-05T07:53:32.249815+00:00"}