{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:TCJM4F2CRM5MC2V6KQ6V4MJ7FY","short_pith_number":"pith:TCJM4F2C","schema_version":"1.0","canonical_sha256":"9892ce17428b3ac16abe543d5e313f2e1358229522013e1319c439b1a62cf7e2","source":{"kind":"arxiv","id":"2409.16578","version":2},"attestation_state":"computed","paper":{"title":"FLaRe: Achieving Masterful and Adaptive Robot Policies with Large-Scale Reinforcement Learning Fine-Tuning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CV","cs.LG"],"primary_cat":"cs.RO","authors_text":"Ali Farhadi, Aniruddha Kembhavi, Jiaheng Hu, Kiana Ehsani, Kuo-Hao Zeng, Peter Stone, Roberto Martin-Martin, Rose Hendrix","submitted_at":"2024-09-25T03:15:17Z","abstract_excerpt":"In recent years, the Robotics field has initiated several efforts toward building generalist robot policies through large-scale multi-task Behavior Cloning. However, direct deployments of these policies have led to unsatisfactory performance, where the policy struggles with unseen states and tasks. How can we break through the performance plateau of these models and elevate their capabilities to new heights? In this paper, we propose FLaRe, a large-scale Reinforcement Learning fine-tuning framework that integrates robust pre-trained representations, large-scale training, and gradient stabiliza"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2409.16578","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.RO","submitted_at":"2024-09-25T03:15:17Z","cross_cats_sorted":["cs.CV","cs.LG"],"title_canon_sha256":"ff7676b14bec8db7b11329704ec9b82cbbd98fa3d21a9e484ad1b5d8b14d696a","abstract_canon_sha256":"b69baf66e6dcf538122ae68a0a95c0686d9faa455204f8e013480f4e56c3072f"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:13:57.231706Z","signature_b64":"8A/8f0D1vAGZaiqkQEutVB/rtbsS73cpIWBC/0GkQbY5Vlbl3iF6e698QmliLcg8GMXld6U4vl8s8hC5n+mSBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"9892ce17428b3ac16abe543d5e313f2e1358229522013e1319c439b1a62cf7e2","last_reissued_at":"2026-07-05T09:13:57.231215Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:13:57.231215Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"FLaRe: Achieving Masterful and Adaptive Robot Policies with Large-Scale Reinforcement Learning Fine-Tuning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CV","cs.LG"],"primary_cat":"cs.RO","authors_text":"Ali Farhadi, Aniruddha Kembhavi, Jiaheng Hu, Kiana Ehsani, Kuo-Hao Zeng, Peter Stone, Roberto Martin-Martin, Rose Hendrix","submitted_at":"2024-09-25T03:15:17Z","abstract_excerpt":"In recent years, the Robotics field has initiated several efforts toward building generalist robot policies through large-scale multi-task Behavior Cloning. However, direct deployments of these policies have led to unsatisfactory performance, where the policy struggles with unseen states and tasks. How can we break through the performance plateau of these models and elevate their capabilities to new heights? In this paper, we propose FLaRe, a large-scale Reinforcement Learning fine-tuning framework that integrates robust pre-trained representations, large-scale training, and gradient stabiliza"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2409.16578","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2409.16578/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2409.16578","created_at":"2026-07-05T09:13:57.231272+00:00"},{"alias_kind":"arxiv_version","alias_value":"2409.16578v2","created_at":"2026-07-05T09:13:57.231272+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2409.16578","created_at":"2026-07-05T09:13:57.231272+00:00"},{"alias_kind":"pith_short_12","alias_value":"TCJM4F2CRM5M","created_at":"2026-07-05T09:13:57.231272+00:00"},{"alias_kind":"pith_short_16","alias_value":"TCJM4F2CRM5MC2V6","created_at":"2026-07-05T09:13:57.231272+00:00"},{"alias_kind":"pith_short_8","alias_value":"TCJM4F2C","created_at":"2026-07-05T09:13:57.231272+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2503.03480","citing_title":"SafeVLA: Towards Safety Alignment of Vision-Language-Action Model via Constrained Learning","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2505.18719","citing_title":"VLA-RL: Towards Masterful and General Robotic Manipulation with Scalable Reinforcement Learning","ref_index":28,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/TCJM4F2CRM5MC2V6KQ6V4MJ7FY","json":"https://pith.science/pith/TCJM4F2CRM5MC2V6KQ6V4MJ7FY.json","graph_json":"https://pith.science/api/pith-number/TCJM4F2CRM5MC2V6KQ6V4MJ7FY/graph.json","events_json":"https://pith.science/api/pith-number/TCJM4F2CRM5MC2V6KQ6V4MJ7FY/events.json","paper":"https://pith.science/paper/TCJM4F2C"},"agent_actions":{"view_html":"https://pith.science/pith/TCJM4F2CRM5MC2V6KQ6V4MJ7FY","download_json":"https://pith.science/pith/TCJM4F2CRM5MC2V6KQ6V4MJ7FY.json","view_paper":"https://pith.science/paper/TCJM4F2C","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2409.16578&json=true","fetch_graph":"https://pith.science/api/pith-number/TCJM4F2CRM5MC2V6KQ6V4MJ7FY/graph.json","fetch_events":"https://pith.science/api/pith-number/TCJM4F2CRM5MC2V6KQ6V4MJ7FY/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/TCJM4F2CRM5MC2V6KQ6V4MJ7FY/action/timestamp_anchor","attest_storage":"https://pith.science/pith/TCJM4F2CRM5MC2V6KQ6V4MJ7FY/action/storage_attestation","attest_author":"https://pith.science/pith/TCJM4F2CRM5MC2V6KQ6V4MJ7FY/action/author_attestation","sign_citation":"https://pith.science/pith/TCJM4F2CRM5MC2V6KQ6V4MJ7FY/action/citation_signature","submit_replication":"https://pith.science/pith/TCJM4F2CRM5MC2V6KQ6V4MJ7FY/action/replication_record"}},"created_at":"2026-07-05T09:13:57.231272+00:00","updated_at":"2026-07-05T09:13:57.231272+00:00"}