{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:KHJIRD3MCPF6D7PBO2S6L2WY4O","short_pith_number":"pith:KHJIRD3M","schema_version":"1.0","canonical_sha256":"51d2888f6c13cbe1fde176a5e5ead8e3a5c2880d88c4cb8bb83387bc768b7acb","source":{"kind":"arxiv","id":"2501.16664","version":1},"attestation_state":"computed","paper":{"title":"Improving Vision-Language-Action Model with Online Reinforcement Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CV","cs.LG"],"primary_cat":"cs.RO","authors_text":"Jianke Zhang, Jianyu Chen, Xiang Ji, Xiaoyu Chen, Yanjiang Guo, Yen-Jen Wang, Yucheng Hu","submitted_at":"2025-01-28T02:53:48Z","abstract_excerpt":"Recent studies have successfully integrated large vision-language models (VLMs) into low-level robotic control by supervised fine-tuning (SFT) with expert robotic datasets, resulting in what we term vision-language-action (VLA) models. Although the VLA models are powerful, how to improve these large models during interaction with environments remains an open question. In this paper, we explore how to further improve these VLA models via Reinforcement Learning (RL), a commonly used fine-tuning technique for large models. However, we find that directly applying online RL to large VLA models pres"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2501.16664","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.RO","submitted_at":"2025-01-28T02:53:48Z","cross_cats_sorted":["cs.CV","cs.LG"],"title_canon_sha256":"761227929c6a2ac5eaf3bd541aa7d2a7c5f328b6a45d7c778d55234625bdf5e4","abstract_canon_sha256":"90d2b34f74ea5545143f131c09b226a57a8ad05cecf83b4826270bfc15d880ff"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:06:16.620055Z","signature_b64":"T7Bgl8G8Dmu6gSfA/BRMV7ZK6tho1PTlICV3YgUUtaZvhVT2f9GWhewP8g/AoC9Cfv09BEiI29SCOOkuUrkiAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"51d2888f6c13cbe1fde176a5e5ead8e3a5c2880d88c4cb8bb83387bc768b7acb","last_reissued_at":"2026-07-05T10:06:16.619602Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:06:16.619602Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Improving Vision-Language-Action Model with Online Reinforcement Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CV","cs.LG"],"primary_cat":"cs.RO","authors_text":"Jianke Zhang, Jianyu Chen, Xiang Ji, Xiaoyu Chen, Yanjiang Guo, Yen-Jen Wang, Yucheng Hu","submitted_at":"2025-01-28T02:53:48Z","abstract_excerpt":"Recent studies have successfully integrated large vision-language models (VLMs) into low-level robotic control by supervised fine-tuning (SFT) with expert robotic datasets, resulting in what we term vision-language-action (VLA) models. Although the VLA models are powerful, how to improve these large models during interaction with environments remains an open question. In this paper, we explore how to further improve these VLA models via Reinforcement Learning (RL), a commonly used fine-tuning technique for large models. However, we find that directly applying online RL to large VLA models pres"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2501.16664","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2501.16664/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2501.16664","created_at":"2026-07-05T10:06:16.619663+00:00"},{"alias_kind":"arxiv_version","alias_value":"2501.16664v1","created_at":"2026-07-05T10:06:16.619663+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2501.16664","created_at":"2026-07-05T10:06:16.619663+00:00"},{"alias_kind":"pith_short_12","alias_value":"KHJIRD3MCPF6","created_at":"2026-07-05T10:06:16.619663+00:00"},{"alias_kind":"pith_short_16","alias_value":"KHJIRD3MCPF6D7PB","created_at":"2026-07-05T10:06:16.619663+00:00"},{"alias_kind":"pith_short_8","alias_value":"KHJIRD3M","created_at":"2026-07-05T10:06:16.619663+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":19,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.23640","citing_title":"Learning Process Rewards via Success Visitation Matching for Efficient RL","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2606.10927","citing_title":"AllDayNav: Lifelong Navigation via Real-World Reinforcement Learning","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2606.31958","citing_title":"Adapting Generalist Robot Policies with Semantic Reinforcement Learning","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2602.07399","citing_title":"VGAS: Value-Guided Action-Chunk Selection for Few-Shot Vision-Language-Action Adaptation","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2605.22896","citing_title":"Agentic-VLA: Efficient Online Adaptation for Vision-Language-Action Models","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2602.10503","citing_title":"Towards Long-Lived Robots: Continual Learning VLA Models via Reinforcement Fine-Tuning","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2602.09023","citing_title":"TwinRL: Digital Twin-Driven Reinforcement Learning for Real-World Robotic Manipulation","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2605.16154","citing_title":"Learn Where Outcomes Diverge: Efficient VLA RL via Probabilistic Chunk Masking","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17486","citing_title":"DyGRO-VLA: Cross-Task Scaling of Vision-Language-Action Models via Dynamic Grouped Residual Optimization","ref_index":133,"is_internal_anchor":false},{"citing_arxiv_id":"2510.12710","citing_title":"Reflection-Based Task Adaptation for Self-Improving VLA","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2505.18719","citing_title":"VLA-RL: Towards Masterful and General Robotic Manipulation with Scalable Reinforcement Learning","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2510.10125","citing_title":"Ctrl-World: A Controllable Generative World Model for Robot Manipulation","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2509.09674","citing_title":"SimpleVLA-RL: Scaling VLA Training via Reinforcement Learning","ref_index":41,"is_internal_anchor":false},{"citing_arxiv_id":"2603.22844","citing_title":"PhySe-RPO: Physics and Semantics Guided Relative Policy Optimization for Diffusion-Based Surgical Smoke Removal","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2502.05855","citing_title":"DexVLA: Vision-Language Model with Plug-In Diffusion Expert for General Robot Control","ref_index":45,"is_internal_anchor":false},{"citing_arxiv_id":"2412.14803","citing_title":"Video Prediction Policy: A Generalist Robot Policy with Predictive Visual Representations","ref_index":101,"is_internal_anchor":false},{"citing_arxiv_id":"2511.14759","citing_title":"$\\pi^{*}_{0.6}$: a VLA That Learns From Experience","ref_index":36,"is_internal_anchor":false},{"citing_arxiv_id":"2604.04502","citing_title":"Veo-Act: How Far Can Frontier Video Models Advance Generalizable Robot Manipulation?","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2604.04834","citing_title":"E-VLA: Event-Augmented Vision-Language-Action Model for Dark and Blurred Scenes","ref_index":23,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/KHJIRD3MCPF6D7PBO2S6L2WY4O","json":"https://pith.science/pith/KHJIRD3MCPF6D7PBO2S6L2WY4O.json","graph_json":"https://pith.science/api/pith-number/KHJIRD3MCPF6D7PBO2S6L2WY4O/graph.json","events_json":"https://pith.science/api/pith-number/KHJIRD3MCPF6D7PBO2S6L2WY4O/events.json","paper":"https://pith.science/paper/KHJIRD3M"},"agent_actions":{"view_html":"https://pith.science/pith/KHJIRD3MCPF6D7PBO2S6L2WY4O","download_json":"https://pith.science/pith/KHJIRD3MCPF6D7PBO2S6L2WY4O.json","view_paper":"https://pith.science/paper/KHJIRD3M","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2501.16664&json=true","fetch_graph":"https://pith.science/api/pith-number/KHJIRD3MCPF6D7PBO2S6L2WY4O/graph.json","fetch_events":"https://pith.science/api/pith-number/KHJIRD3MCPF6D7PBO2S6L2WY4O/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/KHJIRD3MCPF6D7PBO2S6L2WY4O/action/timestamp_anchor","attest_storage":"https://pith.science/pith/KHJIRD3MCPF6D7PBO2S6L2WY4O/action/storage_attestation","attest_author":"https://pith.science/pith/KHJIRD3MCPF6D7PBO2S6L2WY4O/action/author_attestation","sign_citation":"https://pith.science/pith/KHJIRD3MCPF6D7PBO2S6L2WY4O/action/citation_signature","submit_replication":"https://pith.science/pith/KHJIRD3MCPF6D7PBO2S6L2WY4O/action/replication_record"}},"created_at":"2026-07-05T10:06:16.619663+00:00","updated_at":"2026-07-05T10:06:16.619663+00:00"}