{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:BL74AFZ243J6THCSCPCHRDONTB","short_pith_number":"pith:BL74AFZ2","schema_version":"1.0","canonical_sha256":"0affc0173ae6d3e99c5213c4788dcd9851fd75c090c40119f83b4bfa8fcf84cf","source":{"kind":"arxiv","id":"2412.09858","version":1},"attestation_state":"computed","paper":{"title":"RLDG: Robotic Generalist Policy Distillation via Reinforcement Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.RO","authors_text":"Charles Xu, Jianlan Luo, Qiyang Li, Sergey Levine","submitted_at":"2024-12-13T04:57:55Z","abstract_excerpt":"Recent advances in robotic foundation models have enabled the development of generalist policies that can adapt to diverse tasks. While these models show impressive flexibility, their performance heavily depends on the quality of their training data. In this work, we propose Reinforcement Learning Distilled Generalists (RLDG), a method that leverages reinforcement learning to generate high-quality training data for finetuning generalist policies. Through extensive real-world experiments on precise manipulation tasks like connector insertion and assembly, we demonstrate that generalist policies"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2412.09858","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.RO","submitted_at":"2024-12-13T04:57:55Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"27f1fa792f324fdb0f33b208cdd8692b38a1852190adeca60bce9ae443595d98","abstract_canon_sha256":"d6fd22e3bc45d570e970159c9a0ae0a532e46141d47a8396abdaf48d71f113a2"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:48:42.772263Z","signature_b64":"o29rGfUjHgoLXhKazh2z460BpkzooeH0LgMPNyqTf3SX4cygpNcMphFmL4qJTQliCe+BqK6iWcFDYJUwQzwEAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"0affc0173ae6d3e99c5213c4788dcd9851fd75c090c40119f83b4bfa8fcf84cf","last_reissued_at":"2026-07-05T09:48:42.771773Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:48:42.771773Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"RLDG: Robotic Generalist Policy Distillation via Reinforcement Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.RO","authors_text":"Charles Xu, Jianlan Luo, Qiyang Li, Sergey Levine","submitted_at":"2024-12-13T04:57:55Z","abstract_excerpt":"Recent advances in robotic foundation models have enabled the development of generalist policies that can adapt to diverse tasks. While these models show impressive flexibility, their performance heavily depends on the quality of their training data. In this work, we propose Reinforcement Learning Distilled Generalists (RLDG), a method that leverages reinforcement learning to generate high-quality training data for finetuning generalist policies. Through extensive real-world experiments on precise manipulation tasks like connector insertion and assembly, we demonstrate that generalist policies"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2412.09858","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2412.09858/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2412.09858","created_at":"2026-07-05T09:48:42.771832+00:00"},{"alias_kind":"arxiv_version","alias_value":"2412.09858v1","created_at":"2026-07-05T09:48:42.771832+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2412.09858","created_at":"2026-07-05T09:48:42.771832+00:00"},{"alias_kind":"pith_short_12","alias_value":"BL74AFZ243J6","created_at":"2026-07-05T09:48:42.771832+00:00"},{"alias_kind":"pith_short_16","alias_value":"BL74AFZ243J6THCS","created_at":"2026-07-05T09:48:42.771832+00:00"},{"alias_kind":"pith_short_8","alias_value":"BL74AFZ2","created_at":"2026-07-05T09:48:42.771832+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":14,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.24884","citing_title":"InSight: Self-Guided Skill Acquisition via Steerable VLAs","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2606.22471","citing_title":"Scalable Multi-Task Data Generation via Reinforcement Learning for Language-Conditioned Bimanual Dexterous Manipulation","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2606.13675","citing_title":"Improving Robotic Generalist Policies via Flow Reversal Steering","ref_index":48,"is_internal_anchor":false},{"citing_arxiv_id":"2606.10927","citing_title":"AllDayNav: Lifelong Navigation via Real-World Reinforcement Learning","ref_index":62,"is_internal_anchor":false},{"citing_arxiv_id":"2606.09615","citing_title":"DexPIE: Stable Dexterous Policy Improvement from Real-World Experience","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2606.06461","citing_title":"Flow-based Policy Adaptation without Policy Updates","ref_index":47,"is_internal_anchor":false},{"citing_arxiv_id":"2605.00416","citing_title":"Learning While Deploying: Fleet-Scale Reinforcement Learning for Generalist Robot Policies","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2605.25210","citing_title":"Multi-Objective Learning for Diffusion Models: A Statistical Theory under Semi-Supervised Learning","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2606.22471","citing_title":"Scalable Multi-Task Data Generation via Reinforcement Learning for Language-Conditioned Bimanual Dexterous Manipulation","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2606.16776","citing_title":"JoyAI-Sim: A Simulation-Enabled Interconversion Toolchain for the Embodied Data Pyramid","ref_index":48,"is_internal_anchor":false},{"citing_arxiv_id":"2505.18719","citing_title":"VLA-RL: Towards Masterful and General Robotic Manipulation with Scalable Reinforcement Learning","ref_index":78,"is_internal_anchor":false},{"citing_arxiv_id":"2511.14759","citing_title":"$\\pi^{*}_{0.6}$: a VLA That Learns From Experience","ref_index":43,"is_internal_anchor":false},{"citing_arxiv_id":"2605.00416","citing_title":"Learning While Deploying: Fleet-Scale Reinforcement Learning for Generalist Robot Policies","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2604.15483","citing_title":"${\\pi}_{0.7}$: a Steerable Generalist Robotic Foundation Model with Emergent Capabilities","ref_index":53,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/BL74AFZ243J6THCSCPCHRDONTB","json":"https://pith.science/pith/BL74AFZ243J6THCSCPCHRDONTB.json","graph_json":"https://pith.science/api/pith-number/BL74AFZ243J6THCSCPCHRDONTB/graph.json","events_json":"https://pith.science/api/pith-number/BL74AFZ243J6THCSCPCHRDONTB/events.json","paper":"https://pith.science/paper/BL74AFZ2"},"agent_actions":{"view_html":"https://pith.science/pith/BL74AFZ243J6THCSCPCHRDONTB","download_json":"https://pith.science/pith/BL74AFZ243J6THCSCPCHRDONTB.json","view_paper":"https://pith.science/paper/BL74AFZ2","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2412.09858&json=true","fetch_graph":"https://pith.science/api/pith-number/BL74AFZ243J6THCSCPCHRDONTB/graph.json","fetch_events":"https://pith.science/api/pith-number/BL74AFZ243J6THCSCPCHRDONTB/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/BL74AFZ243J6THCSCPCHRDONTB/action/timestamp_anchor","attest_storage":"https://pith.science/pith/BL74AFZ243J6THCSCPCHRDONTB/action/storage_attestation","attest_author":"https://pith.science/pith/BL74AFZ243J6THCSCPCHRDONTB/action/author_attestation","sign_citation":"https://pith.science/pith/BL74AFZ243J6THCSCPCHRDONTB/action/citation_signature","submit_replication":"https://pith.science/pith/BL74AFZ243J6THCSCPCHRDONTB/action/replication_record"}},"created_at":"2026-07-05T09:48:42.771832+00:00","updated_at":"2026-07-05T09:48:42.771832+00:00"}