{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:PRWSEKXDNOW6BGQC6NYCKSUVM2","short_pith_number":"pith:PRWSEKXD","schema_version":"1.0","canonical_sha256":"7c6d222ae36bade09a02f370254a9566829532129d16d732013b22217ccc2f85","source":{"kind":"arxiv","id":"2312.09187","version":3},"attestation_state":"computed","paper":{"title":"Vision-Language Models as a Source of Rewards","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Alexander Neitz, Clare Lyle, Dan Horgan, Dmitry Nikulin, Fabio Pardo, Feryal Behbahani, Gheorghe Comanici, Harris Chan, Himanshu Sahni, Hussain Masoom, Jack Parker-Holder, John Quan, Kate Baumli, Kay McKinney, Kristian Holsheimer, Lei Zhang, Luyu Wang, Maxime Gazeau, Michael Laskin, Richie Steigerwald, Satinder Baveja, Sebastian Flennerhag, Stephen Spencer, Tim Rockt\\\"aschel, Tom Schaul, Volodymyr Mnih, Yannick Schroecker","submitted_at":"2023-12-14T18:06:17Z","abstract_excerpt":"Building generalist agents that can accomplish many goals in rich open-ended environments is one of the research frontiers for reinforcement learning. A key limiting factor for building generalist agents with RL has been the need for a large number of reward functions for achieving different goals. We investigate the feasibility of using off-the-shelf vision-language models, or VLMs, as sources of rewards for reinforcement learning agents. We show how rewards for visual achievement of a variety of language goals can be derived from the CLIP family of models, and used to train RL agents that ca"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2312.09187","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2023-12-14T18:06:17Z","cross_cats_sorted":[],"title_canon_sha256":"6b8114e76a090bc5ddcc2263971c898c948f7d11462f6ba57da048cae4fc3a45","abstract_canon_sha256":"566b04f8c8a14ff1c8a9cce053add86dcbde597fcec0f9b8fb729696068226b6"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:43:24.951638Z","signature_b64":"0sm/YUACzY1ukVhr7trXgVQWTuDPA6UJV97OYnKJfX1Ndwc+7RI/ne6VjOsWVCiSL9GT0MndcMaDiExRr1OUBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"7c6d222ae36bade09a02f370254a9566829532129d16d732013b22217ccc2f85","last_reissued_at":"2026-07-05T08:43:24.951134Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:43:24.951134Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Vision-Language Models as a Source of Rewards","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Alexander Neitz, Clare Lyle, Dan Horgan, Dmitry Nikulin, Fabio Pardo, Feryal Behbahani, Gheorghe Comanici, Harris Chan, Himanshu Sahni, Hussain Masoom, Jack Parker-Holder, John Quan, Kate Baumli, Kay McKinney, Kristian Holsheimer, Lei Zhang, Luyu Wang, Maxime Gazeau, Michael Laskin, Richie Steigerwald, Satinder Baveja, Sebastian Flennerhag, Stephen Spencer, Tim Rockt\\\"aschel, Tom Schaul, Volodymyr Mnih, Yannick Schroecker","submitted_at":"2023-12-14T18:06:17Z","abstract_excerpt":"Building generalist agents that can accomplish many goals in rich open-ended environments is one of the research frontiers for reinforcement learning. A key limiting factor for building generalist agents with RL has been the need for a large number of reward functions for achieving different goals. We investigate the feasibility of using off-the-shelf vision-language models, or VLMs, as sources of rewards for reinforcement learning agents. We show how rewards for visual achievement of a variety of language goals can be derived from the CLIP family of models, and used to train RL agents that ca"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2312.09187","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2312.09187/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2312.09187","created_at":"2026-07-05T08:43:24.951191+00:00"},{"alias_kind":"arxiv_version","alias_value":"2312.09187v3","created_at":"2026-07-05T08:43:24.951191+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2312.09187","created_at":"2026-07-05T08:43:24.951191+00:00"},{"alias_kind":"pith_short_12","alias_value":"PRWSEKXDNOW6","created_at":"2026-07-05T08:43:24.951191+00:00"},{"alias_kind":"pith_short_16","alias_value":"PRWSEKXDNOW6BGQC","created_at":"2026-07-05T08:43:24.951191+00:00"},{"alias_kind":"pith_short_8","alias_value":"PRWSEKXD","created_at":"2026-07-05T08:43:24.951191+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.23640","citing_title":"Learning Process Rewards via Success Visitation Matching for Efficient RL","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2606.32034","citing_title":"QVal: Cheaply Evaluating Dense Supervision Signals for Long-Horizon LLM Agents","ref_index":45,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/PRWSEKXDNOW6BGQC6NYCKSUVM2","json":"https://pith.science/pith/PRWSEKXDNOW6BGQC6NYCKSUVM2.json","graph_json":"https://pith.science/api/pith-number/PRWSEKXDNOW6BGQC6NYCKSUVM2/graph.json","events_json":"https://pith.science/api/pith-number/PRWSEKXDNOW6BGQC6NYCKSUVM2/events.json","paper":"https://pith.science/paper/PRWSEKXD"},"agent_actions":{"view_html":"https://pith.science/pith/PRWSEKXDNOW6BGQC6NYCKSUVM2","download_json":"https://pith.science/pith/PRWSEKXDNOW6BGQC6NYCKSUVM2.json","view_paper":"https://pith.science/paper/PRWSEKXD","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2312.09187&json=true","fetch_graph":"https://pith.science/api/pith-number/PRWSEKXDNOW6BGQC6NYCKSUVM2/graph.json","fetch_events":"https://pith.science/api/pith-number/PRWSEKXDNOW6BGQC6NYCKSUVM2/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/PRWSEKXDNOW6BGQC6NYCKSUVM2/action/timestamp_anchor","attest_storage":"https://pith.science/pith/PRWSEKXDNOW6BGQC6NYCKSUVM2/action/storage_attestation","attest_author":"https://pith.science/pith/PRWSEKXDNOW6BGQC6NYCKSUVM2/action/author_attestation","sign_citation":"https://pith.science/pith/PRWSEKXDNOW6BGQC6NYCKSUVM2/action/citation_signature","submit_replication":"https://pith.science/pith/PRWSEKXDNOW6BGQC6NYCKSUVM2/action/replication_record"}},"created_at":"2026-07-05T08:43:24.951191+00:00","updated_at":"2026-07-05T08:43:24.951191+00:00"}