{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:IIBU7NEKIIHSLK4NTGVYOGLFOE","short_pith_number":"pith:IIBU7NEK","schema_version":"1.0","canonical_sha256":"42034fb48a420f25ab8d99ab871965711eee473deba295cced71455a28c84056","source":{"kind":"arxiv","id":"2410.04734","version":2},"attestation_state":"computed","paper":{"title":"TLDR: Token-Level Detective Reward Model for Large Vision Language Models","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.CL","cs.CV"],"primary_cat":"cs.LG","authors_text":"Deqing Fu, Guan Pang, Lawrence Chen, Pengchuan Zhang, Robin Jia, Rui Wang, Tong Xiao, Wang Zhu","submitted_at":"2024-10-07T04:00:22Z","abstract_excerpt":"Although reward models have been successful in improving multimodal large language models, the reward models themselves remain brutal and contain minimal information. Notably, existing reward models only mimic human annotations by assigning only one binary feedback to any text, no matter how long the text is. In the realm of multimodal language models, where models are required to process both images and texts, a naive reward model may learn implicit biases toward texts and become less grounded in images. In this paper, we propose a $\\textbf{T}$oken-$\\textbf{L}$evel $\\textbf{D}$etective $\\text"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.04734","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.LG","submitted_at":"2024-10-07T04:00:22Z","cross_cats_sorted":["cs.CL","cs.CV"],"title_canon_sha256":"b8d6ebb4d630dd96320c2b223d7089816d271f2c2167363f5568bfeb1b67b43c","abstract_canon_sha256":"28eea45c2f9f356381b0d2e07bccb76546a372b50172e0819a4e4fad4aa3023a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:19:29.234570Z","signature_b64":"kh7WA4a1Vl267646aVg5QIryTvvTTSEOG1Xc/HhOxKA4kIpHA2wx395IUYA2bDMFg4MqeLIsfMrieN/ecmUaCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"42034fb48a420f25ab8d99ab871965711eee473deba295cced71455a28c84056","last_reissued_at":"2026-07-05T10:19:29.234068Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:19:29.234068Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"TLDR: Token-Level Detective Reward Model for Large Vision Language Models","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.CL","cs.CV"],"primary_cat":"cs.LG","authors_text":"Deqing Fu, Guan Pang, Lawrence Chen, Pengchuan Zhang, Robin Jia, Rui Wang, Tong Xiao, Wang Zhu","submitted_at":"2024-10-07T04:00:22Z","abstract_excerpt":"Although reward models have been successful in improving multimodal large language models, the reward models themselves remain brutal and contain minimal information. Notably, existing reward models only mimic human annotations by assigning only one binary feedback to any text, no matter how long the text is. In the realm of multimodal language models, where models are required to process both images and texts, a naive reward model may learn implicit biases toward texts and become less grounded in images. In this paper, we propose a $\\textbf{T}$oken-$\\textbf{L}$evel $\\textbf{D}$etective $\\text"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.04734","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.04734/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.04734","created_at":"2026-07-05T10:19:29.234127+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.04734v2","created_at":"2026-07-05T10:19:29.234127+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.04734","created_at":"2026-07-05T10:19:29.234127+00:00"},{"alias_kind":"pith_short_12","alias_value":"IIBU7NEKIIHS","created_at":"2026-07-05T10:19:29.234127+00:00"},{"alias_kind":"pith_short_16","alias_value":"IIBU7NEKIIHSLK4N","created_at":"2026-07-05T10:19:29.234127+00:00"},{"alias_kind":"pith_short_8","alias_value":"IIBU7NEK","created_at":"2026-07-05T10:19:29.234127+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/IIBU7NEKIIHSLK4NTGVYOGLFOE","json":"https://pith.science/pith/IIBU7NEKIIHSLK4NTGVYOGLFOE.json","graph_json":"https://pith.science/api/pith-number/IIBU7NEKIIHSLK4NTGVYOGLFOE/graph.json","events_json":"https://pith.science/api/pith-number/IIBU7NEKIIHSLK4NTGVYOGLFOE/events.json","paper":"https://pith.science/paper/IIBU7NEK"},"agent_actions":{"view_html":"https://pith.science/pith/IIBU7NEKIIHSLK4NTGVYOGLFOE","download_json":"https://pith.science/pith/IIBU7NEKIIHSLK4NTGVYOGLFOE.json","view_paper":"https://pith.science/paper/IIBU7NEK","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.04734&json=true","fetch_graph":"https://pith.science/api/pith-number/IIBU7NEKIIHSLK4NTGVYOGLFOE/graph.json","fetch_events":"https://pith.science/api/pith-number/IIBU7NEKIIHSLK4NTGVYOGLFOE/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/IIBU7NEKIIHSLK4NTGVYOGLFOE/action/timestamp_anchor","attest_storage":"https://pith.science/pith/IIBU7NEKIIHSLK4NTGVYOGLFOE/action/storage_attestation","attest_author":"https://pith.science/pith/IIBU7NEKIIHSLK4NTGVYOGLFOE/action/author_attestation","sign_citation":"https://pith.science/pith/IIBU7NEKIIHSLK4NTGVYOGLFOE/action/citation_signature","submit_replication":"https://pith.science/pith/IIBU7NEKIIHSLK4NTGVYOGLFOE/action/replication_record"}},"created_at":"2026-07-05T10:19:29.234127+00:00","updated_at":"2026-07-05T10:19:29.234127+00:00"}