{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:J5535U7K5BUR7BOO5QBRF2FI3P","short_pith_number":"pith:J5535U7K","schema_version":"1.0","canonical_sha256":"4f7bbed3eae8691f85ceec0312e8a8dbdd00336e6f73187e1a7e856df14762e2","source":{"kind":"arxiv","id":"2502.01616","version":1},"attestation_state":"computed","paper":{"title":"Preference VLM: Leveraging VLMs for Scalable Preference-Based Reinforcement Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Amit Roy-Chowdhury, Dripta S. Raychaudhuri, Jiachen Li, Konstantinos Karydis, Udita Ghosh","submitted_at":"2025-02-03T18:50:15Z","abstract_excerpt":"Preference-based reinforcement learning (RL) offers a promising approach for aligning policies with human intent but is often constrained by the high cost of human feedback. In this work, we introduce PrefVLM, a framework that integrates Vision-Language Models (VLMs) with selective human feedback to significantly reduce annotation requirements while maintaining performance. Our method leverages VLMs to generate initial preference labels, which are then filtered to identify uncertain cases for targeted human annotation. Additionally, we adapt VLMs using a self-supervised inverse dynamics loss t"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.01616","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-02-03T18:50:15Z","cross_cats_sorted":[],"title_canon_sha256":"14934472996a1cb8ab2fbeaba9f13659f533ca5b0fc98c0160d9956c9b16d974","abstract_canon_sha256":"7c8ed243d31bc6a627bf658397bb812c900773c0e129dca0f131c32134e08f55"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:08:57.043348Z","signature_b64":"vCBtKVJb4gJACHrIfZfdJvQ3eYhFLhvucE56uxRVirWnbHchfIW2VajA+giKicYZdGOztq1eIIo4/e/m/XpfCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"4f7bbed3eae8691f85ceec0312e8a8dbdd00336e6f73187e1a7e856df14762e2","last_reissued_at":"2026-07-05T10:08:57.042844Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:08:57.042844Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Preference VLM: Leveraging VLMs for Scalable Preference-Based Reinforcement Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Amit Roy-Chowdhury, Dripta S. Raychaudhuri, Jiachen Li, Konstantinos Karydis, Udita Ghosh","submitted_at":"2025-02-03T18:50:15Z","abstract_excerpt":"Preference-based reinforcement learning (RL) offers a promising approach for aligning policies with human intent but is often constrained by the high cost of human feedback. In this work, we introduce PrefVLM, a framework that integrates Vision-Language Models (VLMs) with selective human feedback to significantly reduce annotation requirements while maintaining performance. Our method leverages VLMs to generate initial preference labels, which are then filtered to identify uncertain cases for targeted human annotation. Additionally, we adapt VLMs using a self-supervised inverse dynamics loss t"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.01616","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.01616/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.01616","created_at":"2026-07-05T10:08:57.042901+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.01616v1","created_at":"2026-07-05T10:08:57.042901+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.01616","created_at":"2026-07-05T10:08:57.042901+00:00"},{"alias_kind":"pith_short_12","alias_value":"J5535U7K5BUR","created_at":"2026-07-05T10:08:57.042901+00:00"},{"alias_kind":"pith_short_16","alias_value":"J5535U7K5BUR7BOO","created_at":"2026-07-05T10:08:57.042901+00:00"},{"alias_kind":"pith_short_8","alias_value":"J5535U7K","created_at":"2026-07-05T10:08:57.042901+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2607.01721","citing_title":"CoRe: Combined Rewards with Vision-Language Model Feedback for Preference-Aligned Reinforcement Learning","ref_index":83,"is_internal_anchor":false},{"citing_arxiv_id":"2605.22123","citing_title":"Beyond Pixels: Learning Invariant Rewards for Real-World Robotics From a Few Demonstrations","ref_index":27,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/J5535U7K5BUR7BOO5QBRF2FI3P","json":"https://pith.science/pith/J5535U7K5BUR7BOO5QBRF2FI3P.json","graph_json":"https://pith.science/api/pith-number/J5535U7K5BUR7BOO5QBRF2FI3P/graph.json","events_json":"https://pith.science/api/pith-number/J5535U7K5BUR7BOO5QBRF2FI3P/events.json","paper":"https://pith.science/paper/J5535U7K"},"agent_actions":{"view_html":"https://pith.science/pith/J5535U7K5BUR7BOO5QBRF2FI3P","download_json":"https://pith.science/pith/J5535U7K5BUR7BOO5QBRF2FI3P.json","view_paper":"https://pith.science/paper/J5535U7K","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.01616&json=true","fetch_graph":"https://pith.science/api/pith-number/J5535U7K5BUR7BOO5QBRF2FI3P/graph.json","fetch_events":"https://pith.science/api/pith-number/J5535U7K5BUR7BOO5QBRF2FI3P/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/J5535U7K5BUR7BOO5QBRF2FI3P/action/timestamp_anchor","attest_storage":"https://pith.science/pith/J5535U7K5BUR7BOO5QBRF2FI3P/action/storage_attestation","attest_author":"https://pith.science/pith/J5535U7K5BUR7BOO5QBRF2FI3P/action/author_attestation","sign_citation":"https://pith.science/pith/J5535U7K5BUR7BOO5QBRF2FI3P/action/citation_signature","submit_replication":"https://pith.science/pith/J5535U7K5BUR7BOO5QBRF2FI3P/action/replication_record"}},"created_at":"2026-07-05T10:08:57.042901+00:00","updated_at":"2026-07-05T10:08:57.042901+00:00"}