{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2024:SFYXHPRHYVRL2T7MPMZ5EN5OIZ","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"5c36dfe643a6a0ccd8ec27d837125088d8b045353def7f24fc3232b88e40459c","cross_cats_sorted":["cs.CL"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-08-22T03:49:18Z","title_canon_sha256":"7bc166fd2a1eaf275cd78ed055d2a8118a7e40d67d291800a00d645ab932dc5b"},"schema_version":"1.0","source":{"id":"2408.12109","kind":"arxiv","version":2}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2408.12109","created_at":"2026-07-05T10:07:43Z"},{"alias_kind":"arxiv_version","alias_value":"2408.12109v2","created_at":"2026-07-05T10:07:43Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2408.12109","created_at":"2026-07-05T10:07:43Z"},{"alias_kind":"pith_short_12","alias_value":"SFYXHPRHYVRL","created_at":"2026-07-05T10:07:43Z"},{"alias_kind":"pith_short_16","alias_value":"SFYXHPRHYVRL2T7M","created_at":"2026-07-05T10:07:43Z"},{"alias_kind":"pith_short_8","alias_value":"SFYXHPRH","created_at":"2026-07-05T10:07:43Z"}],"graph_snapshots":[{"event_id":"sha256:8cc7edd4d8110843da4def81b80243cfe2d608ecd74510cb89dad2df505e6710","target":"graph","created_at":"2026-07-05T10:07:43Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2408.12109/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Large vision-language models (LVLMs) often fail to align with human preferences, leading to issues like generating misleading content without proper visual context (also known as hallucination). A promising solution to this problem is using human-preference alignment techniques, such as best-of-n sampling and reinforcement learning. However, these techniques face the difficulty arising from the scarcity of visual preference data, which is required to train a visual reward model (VRM). In this work, we continue the line of research. We present a Robust Visual Reward Model (RoVRM) which improves","authors_text":"Chenglong Wang, Chunliang Zhang, Di Yang, Jingbo Zhu, Murun Yang, Qiaozhi He, Quan Du, Tongran Liu, Tong Xiao, Yang Gan, Yifu Huo, Yongyu Mu","cross_cats":["cs.CL"],"headline":"","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-08-22T03:49:18Z","title":"RoVRM: A Robust Visual Reward Model Optimized via Auxiliary Textual Preference Data"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2408.12109","kind":"arxiv","version":2},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:891764df6233521c4c823cd87e5237cc3ef51f9fd6c5045289fd992a8238c90b","target":"record","created_at":"2026-07-05T10:07:43Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"5c36dfe643a6a0ccd8ec27d837125088d8b045353def7f24fc3232b88e40459c","cross_cats_sorted":["cs.CL"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2024-08-22T03:49:18Z","title_canon_sha256":"7bc166fd2a1eaf275cd78ed055d2a8118a7e40d67d291800a00d645ab932dc5b"},"schema_version":"1.0","source":{"id":"2408.12109","kind":"arxiv","version":2}},"canonical_sha256":"917173be27c562bd4fec7b33d237ae466d53c40125cf8895a89e51cf44899395","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"917173be27c562bd4fec7b33d237ae466d53c40125cf8895a89e51cf44899395","first_computed_at":"2026-07-05T10:07:43.733447Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T10:07:43.733447Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"M2BAGa/ckifCmS2Re7E3OewMgPRJwzlwcLYcQJ822/sxaEHEMYKF/NgGaBxBJTe+bnyAC7PtGkuA66iBzGH4Cw==","signature_status":"signed_v1","signed_at":"2026-07-05T10:07:43.733946Z","signed_message":"canonical_sha256_bytes"},"source_id":"2408.12109","source_kind":"arxiv","source_version":2}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:891764df6233521c4c823cd87e5237cc3ef51f9fd6c5045289fd992a8238c90b","sha256:8cc7edd4d8110843da4def81b80243cfe2d608ecd74510cb89dad2df505e6710"],"state_sha256":"2c0a7e84fa4a2733c561692de57845503d5a45068f54280c3cfd695355f1a8f4"}