{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:Z6YO36QGDXQ3SF7SRQ6Q7S7LCT","short_pith_number":"pith:Z6YO36QG","schema_version":"1.0","canonical_sha256":"cfb0edfa061de1b917f28c3d0fcbeb14e3c7ee5e6ce9e53c2fe96c8e777364d4","source":{"kind":"arxiv","id":"2402.03681","version":4},"attestation_state":"computed","paper":{"title":"RL-VLM-F: Reinforcement Learning from Vision Language Foundation Model Feedback","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.RO","authors_text":"David Held, Erdem Biyik, Jesse Zhang, Yufei Wang, Zackory Erickson, Zhanyi Sun, Zhou Xian","submitted_at":"2024-02-06T04:06:06Z","abstract_excerpt":"Reward engineering has long been a challenge in Reinforcement Learning (RL) research, as it often requires extensive human effort and iterative processes of trial-and-error to design effective reward functions. In this paper, we propose RL-VLM-F, a method that automatically generates reward functions for agents to learn new tasks, using only a text description of the task goal and the agent's visual observations, by leveraging feedbacks from vision language foundation models (VLMs). The key to our approach is to query these models to give preferences over pairs of the agent's image observation"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2402.03681","kind":"arxiv","version":4},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.RO","submitted_at":"2024-02-06T04:06:06Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"43358ee52b45376b1694921b8292162ffbf00457391edc188c6f87a638a9a6fd","abstract_canon_sha256":"07d675c3f34ba20997781e7741ff682f02fdf46649431a6f258dad6e31d39826"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:32:11.070211Z","signature_b64":"UV83gJSYuTnQHu543CwL5SPhJkIOCXCDulskK66UUUlYwjiVzOcjjftiCYu7wcbAv6X+58QD7FJLVLQHN5SFAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"cfb0edfa061de1b917f28c3d0fcbeb14e3c7ee5e6ce9e53c2fe96c8e777364d4","last_reissued_at":"2026-07-05T08:32:11.069707Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:32:11.069707Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"RL-VLM-F: Reinforcement Learning from Vision Language Foundation Model Feedback","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.RO","authors_text":"David Held, Erdem Biyik, Jesse Zhang, Yufei Wang, Zackory Erickson, Zhanyi Sun, Zhou Xian","submitted_at":"2024-02-06T04:06:06Z","abstract_excerpt":"Reward engineering has long been a challenge in Reinforcement Learning (RL) research, as it often requires extensive human effort and iterative processes of trial-and-error to design effective reward functions. In this paper, we propose RL-VLM-F, a method that automatically generates reward functions for agents to learn new tasks, using only a text description of the task goal and the agent's visual observations, by leveraging feedbacks from vision language foundation models (VLMs). The key to our approach is to query these models to give preferences over pairs of the agent's image observation"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2402.03681","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2402.03681/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2402.03681","created_at":"2026-07-05T08:32:11.069769+00:00"},{"alias_kind":"arxiv_version","alias_value":"2402.03681v4","created_at":"2026-07-05T08:32:11.069769+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2402.03681","created_at":"2026-07-05T08:32:11.069769+00:00"},{"alias_kind":"pith_short_12","alias_value":"Z6YO36QGDXQ3","created_at":"2026-07-05T08:32:11.069769+00:00"},{"alias_kind":"pith_short_16","alias_value":"Z6YO36QGDXQ3SF7S","created_at":"2026-07-05T08:32:11.069769+00:00"},{"alias_kind":"pith_short_8","alias_value":"Z6YO36QG","created_at":"2026-07-05T08:32:11.069769+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":15,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.05391","citing_title":"LLM-as-a-Verifier: A General-Purpose Verification Framework","ref_index":80,"is_internal_anchor":true},{"citing_arxiv_id":"2606.25398","citing_title":"MAPL: Multi-Objective Preference Learning for Robot Locomotion","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2606.23640","citing_title":"Learning Process Rewards via Success Visitation Matching for Efficient RL","ref_index":89,"is_internal_anchor":false},{"citing_arxiv_id":"2606.12072","citing_title":"World Model Self-Distillation: Training World Models to Solve General Tasks","ref_index":65,"is_internal_anchor":false},{"citing_arxiv_id":"2606.20641","citing_title":"MAGNIFIED: RL Fine-tuning of Multimodal Large Language Models for Motion Planning","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2606.28570","citing_title":"Digitizing Coaching Intelligence: An Agentic Framework for Holistic Athlete Profiling using VLM and RAG","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2606.32027","citing_title":"Freeform Preference Learning for Robotic Manipulation","ref_index":42,"is_internal_anchor":false},{"citing_arxiv_id":"2606.32034","citing_title":"QVal: Cheaply Evaluating Dense Supervision Signals for Long-Horizon LLM Agents","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2606.00931","citing_title":"CV-Arena: An Open Benchmark for Instructional Computer Vision Problem Solving with Human-AI Collaborative Preferences","ref_index":71,"is_internal_anchor":false},{"citing_arxiv_id":"2605.22123","citing_title":"Beyond Pixels: Learning Invariant Rewards for Real-World Robotics From a Few Demonstrations","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18903","citing_title":"Reasoning Portability: Guiding Continual Learning for MLLMs in the RLVR Era","ref_index":52,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17933","citing_title":"AtlasVA: Self-Evolving Visual Skill Memory for Teacher-Free VLM Agents","ref_index":43,"is_internal_anchor":false},{"citing_arxiv_id":"2508.11196","citing_title":"UAV-VL-R1: Generalizing Vision-Language Models via Supervised Fine-Tuning and Multi-Stage GRPO for UAV Visual Reasoning","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2510.12710","citing_title":"Reflection-Based Task Adaptation for Self-Improving VLA","ref_index":39,"is_internal_anchor":false},{"citing_arxiv_id":"2602.19837","citing_title":"Meta-Learning and Meta-Reinforcement Learning -- Tracing the Path towards DeepMind's Adaptive Agent","ref_index":182,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/Z6YO36QGDXQ3SF7SRQ6Q7S7LCT","json":"https://pith.science/pith/Z6YO36QGDXQ3SF7SRQ6Q7S7LCT.json","graph_json":"https://pith.science/api/pith-number/Z6YO36QGDXQ3SF7SRQ6Q7S7LCT/graph.json","events_json":"https://pith.science/api/pith-number/Z6YO36QGDXQ3SF7SRQ6Q7S7LCT/events.json","paper":"https://pith.science/paper/Z6YO36QG"},"agent_actions":{"view_html":"https://pith.science/pith/Z6YO36QGDXQ3SF7SRQ6Q7S7LCT","download_json":"https://pith.science/pith/Z6YO36QGDXQ3SF7SRQ6Q7S7LCT.json","view_paper":"https://pith.science/paper/Z6YO36QG","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2402.03681&json=true","fetch_graph":"https://pith.science/api/pith-number/Z6YO36QGDXQ3SF7SRQ6Q7S7LCT/graph.json","fetch_events":"https://pith.science/api/pith-number/Z6YO36QGDXQ3SF7SRQ6Q7S7LCT/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/Z6YO36QGDXQ3SF7SRQ6Q7S7LCT/action/timestamp_anchor","attest_storage":"https://pith.science/pith/Z6YO36QGDXQ3SF7SRQ6Q7S7LCT/action/storage_attestation","attest_author":"https://pith.science/pith/Z6YO36QGDXQ3SF7SRQ6Q7S7LCT/action/author_attestation","sign_citation":"https://pith.science/pith/Z6YO36QGDXQ3SF7SRQ6Q7S7LCT/action/citation_signature","submit_replication":"https://pith.science/pith/Z6YO36QGDXQ3SF7SRQ6Q7S7LCT/action/replication_record"}},"created_at":"2026-07-05T08:32:11.069769+00:00","updated_at":"2026-07-05T08:32:11.069769+00:00"}