{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:IRVHYJEGCVKQCV7L5ZR6NMSYS7","short_pith_number":"pith:IRVHYJEG","schema_version":"1.0","canonical_sha256":"446a7c248615550157ebee63e6b25897d3cf2ea9a1111349003f1b2993d13e5e","source":{"kind":"arxiv","id":"2508.11616","version":1},"attestation_state":"computed","paper":{"title":"Controlling Multimodal LLMs via Reward-guided Decoding","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.LG"],"primary_cat":"cs.CV","authors_text":"Adriana Romero-Soriano, Aishwarya Agrawal, Koustuv Sinha, Michal Drozdzal, Oscar Ma\\~nas, Pierluca D'Oro","submitted_at":"2025-08-15T17:29:06Z","abstract_excerpt":"As Multimodal Large Language Models (MLLMs) gain widespread applicability, it is becoming increasingly desirable to adapt them for diverse user needs. In this paper, we study the adaptation of MLLMs through controlled decoding. To achieve this, we introduce the first method for reward-guided decoding of MLLMs and demonstrate its application in improving their visual grounding. Our method involves building reward models for visual grounding and using them to guide the MLLM's decoding process. Concretely, we build two separate reward models to independently control the degree of object precision"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2508.11616","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2025-08-15T17:29:06Z","cross_cats_sorted":["cs.AI","cs.CL","cs.LG"],"title_canon_sha256":"16c499c92fb4a8bdc7b28e5a6bd1d6ce227c7e855c97e892cb807aee8a84b376","abstract_canon_sha256":"94808666d91d5569a32a181c5cd50b40838d2641dd141513285614731a2a3cd7"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:54:36.272273Z","signature_b64":"K1cwKDfLPd8bWDD73x2PU+Ak0iBpTPvzgJe82czYGA5ZOd73rOmZBT/E/rzpUhAaYjXgcTRa1hVPubNa8/KoCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"446a7c248615550157ebee63e6b25897d3cf2ea9a1111349003f1b2993d13e5e","last_reissued_at":"2026-07-05T11:54:36.271803Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:54:36.271803Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Controlling Multimodal LLMs via Reward-guided Decoding","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.LG"],"primary_cat":"cs.CV","authors_text":"Adriana Romero-Soriano, Aishwarya Agrawal, Koustuv Sinha, Michal Drozdzal, Oscar Ma\\~nas, Pierluca D'Oro","submitted_at":"2025-08-15T17:29:06Z","abstract_excerpt":"As Multimodal Large Language Models (MLLMs) gain widespread applicability, it is becoming increasingly desirable to adapt them for diverse user needs. In this paper, we study the adaptation of MLLMs through controlled decoding. To achieve this, we introduce the first method for reward-guided decoding of MLLMs and demonstrate its application in improving their visual grounding. Our method involves building reward models for visual grounding and using them to guide the MLLM's decoding process. Concretely, we build two separate reward models to independently control the degree of object precision"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2508.11616","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2508.11616/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2508.11616","created_at":"2026-07-05T11:54:36.271861+00:00"},{"alias_kind":"arxiv_version","alias_value":"2508.11616v1","created_at":"2026-07-05T11:54:36.271861+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2508.11616","created_at":"2026-07-05T11:54:36.271861+00:00"},{"alias_kind":"pith_short_12","alias_value":"IRVHYJEGCVKQ","created_at":"2026-07-05T11:54:36.271861+00:00"},{"alias_kind":"pith_short_16","alias_value":"IRVHYJEGCVKQCV7L","created_at":"2026-07-05T11:54:36.271861+00:00"},{"alias_kind":"pith_short_8","alias_value":"IRVHYJEG","created_at":"2026-07-05T11:54:36.271861+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/IRVHYJEGCVKQCV7L5ZR6NMSYS7","json":"https://pith.science/pith/IRVHYJEGCVKQCV7L5ZR6NMSYS7.json","graph_json":"https://pith.science/api/pith-number/IRVHYJEGCVKQCV7L5ZR6NMSYS7/graph.json","events_json":"https://pith.science/api/pith-number/IRVHYJEGCVKQCV7L5ZR6NMSYS7/events.json","paper":"https://pith.science/paper/IRVHYJEG"},"agent_actions":{"view_html":"https://pith.science/pith/IRVHYJEGCVKQCV7L5ZR6NMSYS7","download_json":"https://pith.science/pith/IRVHYJEGCVKQCV7L5ZR6NMSYS7.json","view_paper":"https://pith.science/paper/IRVHYJEG","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2508.11616&json=true","fetch_graph":"https://pith.science/api/pith-number/IRVHYJEGCVKQCV7L5ZR6NMSYS7/graph.json","fetch_events":"https://pith.science/api/pith-number/IRVHYJEGCVKQCV7L5ZR6NMSYS7/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/IRVHYJEGCVKQCV7L5ZR6NMSYS7/action/timestamp_anchor","attest_storage":"https://pith.science/pith/IRVHYJEGCVKQCV7L5ZR6NMSYS7/action/storage_attestation","attest_author":"https://pith.science/pith/IRVHYJEGCVKQCV7L5ZR6NMSYS7/action/author_attestation","sign_citation":"https://pith.science/pith/IRVHYJEGCVKQCV7L5ZR6NMSYS7/action/citation_signature","submit_replication":"https://pith.science/pith/IRVHYJEGCVKQCV7L5ZR6NMSYS7/action/replication_record"}},"created_at":"2026-07-05T11:54:36.271861+00:00","updated_at":"2026-07-05T11:54:36.271861+00:00"}