{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:RNQURRIVQADA2LA7UZYXXZMHLD","short_pith_number":"pith:RNQURRIV","schema_version":"1.0","canonical_sha256":"8b6148c51580060d2c1fa6717be58758da2a64ffe578eb01abff65a52cabe8c9","source":{"kind":"arxiv","id":"2302.08242","version":1},"attestation_state":"computed","paper":{"title":"Tuning computer vision models with task rewards","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Alexander Kolesnikov, Andr\\'e Susano Pinto, Lucas Beyer, Xiaohua Zhai, Yuge Shi","submitted_at":"2023-02-16T11:49:48Z","abstract_excerpt":"Misalignment between model predictions and intended usage can be detrimental for the deployment of computer vision models. The issue is exacerbated when the task involves complex structured outputs, as it becomes harder to design procedures which address this misalignment. In natural language processing, this is often addressed using reinforcement learning techniques that align models with a task reward. We adopt this approach and show its surprising effectiveness across multiple computer vision tasks, such as object detection, panoptic segmentation, colorization and image captioning. We belie"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2302.08242","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2023-02-16T11:49:48Z","cross_cats_sorted":[],"title_canon_sha256":"f130d4de3eebf88db4cd35c8dcc1a0bc862854a4922296fbaa4ec24d399ed876","abstract_canon_sha256":"dc80a523ddff1ce4fec82cf970cad7bd74a11376ddf9356d5ea0de744f33f416"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:42:35.982660Z","signature_b64":"IdGZqO2p7JWBTqs43P55IyJIYJ+UMH9n4MBvFpTuJehKtLmwvPoB6/NwtqmjXPF5CrOhcR1WJ0OZMaNYnbflBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"8b6148c51580060d2c1fa6717be58758da2a64ffe578eb01abff65a52cabe8c9","last_reissued_at":"2026-07-05T05:42:35.982181Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:42:35.982181Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Tuning computer vision models with task rewards","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Alexander Kolesnikov, Andr\\'e Susano Pinto, Lucas Beyer, Xiaohua Zhai, Yuge Shi","submitted_at":"2023-02-16T11:49:48Z","abstract_excerpt":"Misalignment between model predictions and intended usage can be detrimental for the deployment of computer vision models. The issue is exacerbated when the task involves complex structured outputs, as it becomes harder to design procedures which address this misalignment. In natural language processing, this is often addressed using reinforcement learning techniques that align models with a task reward. We adopt this approach and show its surprising effectiveness across multiple computer vision tasks, such as object detection, panoptic segmentation, colorization and image captioning. We belie"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2302.08242","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2302.08242/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2302.08242","created_at":"2026-07-05T05:42:35.982237+00:00"},{"alias_kind":"arxiv_version","alias_value":"2302.08242v1","created_at":"2026-07-05T05:42:35.982237+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2302.08242","created_at":"2026-07-05T05:42:35.982237+00:00"},{"alias_kind":"pith_short_12","alias_value":"RNQURRIVQADA","created_at":"2026-07-05T05:42:35.982237+00:00"},{"alias_kind":"pith_short_16","alias_value":"RNQURRIVQADA2LA7","created_at":"2026-07-05T05:42:35.982237+00:00"},{"alias_kind":"pith_short_8","alias_value":"RNQURRIV","created_at":"2026-07-05T05:42:35.982237+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2508.09128","citing_title":"A Review On Safe Reinforcement Learning Using Lyapunov and Barrier Functions","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06987","citing_title":"Response Time Enhances Alignment with Heterogeneous Preferences","ref_index":92,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/RNQURRIVQADA2LA7UZYXXZMHLD","json":"https://pith.science/pith/RNQURRIVQADA2LA7UZYXXZMHLD.json","graph_json":"https://pith.science/api/pith-number/RNQURRIVQADA2LA7UZYXXZMHLD/graph.json","events_json":"https://pith.science/api/pith-number/RNQURRIVQADA2LA7UZYXXZMHLD/events.json","paper":"https://pith.science/paper/RNQURRIV"},"agent_actions":{"view_html":"https://pith.science/pith/RNQURRIVQADA2LA7UZYXXZMHLD","download_json":"https://pith.science/pith/RNQURRIVQADA2LA7UZYXXZMHLD.json","view_paper":"https://pith.science/paper/RNQURRIV","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2302.08242&json=true","fetch_graph":"https://pith.science/api/pith-number/RNQURRIVQADA2LA7UZYXXZMHLD/graph.json","fetch_events":"https://pith.science/api/pith-number/RNQURRIVQADA2LA7UZYXXZMHLD/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/RNQURRIVQADA2LA7UZYXXZMHLD/action/timestamp_anchor","attest_storage":"https://pith.science/pith/RNQURRIVQADA2LA7UZYXXZMHLD/action/storage_attestation","attest_author":"https://pith.science/pith/RNQURRIVQADA2LA7UZYXXZMHLD/action/author_attestation","sign_citation":"https://pith.science/pith/RNQURRIVQADA2LA7UZYXXZMHLD/action/citation_signature","submit_replication":"https://pith.science/pith/RNQURRIVQADA2LA7UZYXXZMHLD/action/replication_record"}},"created_at":"2026-07-05T05:42:35.982237+00:00","updated_at":"2026-07-05T05:42:35.982237+00:00"}