{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:YGMFAJDTLZWN3YNJZVSKIUXMZL","short_pith_number":"pith:YGMFAJDT","schema_version":"1.0","canonical_sha256":"c1985024735e6cdde1a9cd64a452eccad4abac68093b1e52ffdfcdad3a8a2b2a","source":{"kind":"arxiv","id":"2311.02198","version":6},"attestation_state":"computed","paper":{"title":"Imitation Bootstrapped Reinforcement Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Dorsa Sadigh, Hengyuan Hu, Suvir Mirchandani","submitted_at":"2023-11-03T19:03:20Z","abstract_excerpt":"Despite the considerable potential of reinforcement learning (RL), robotic control tasks predominantly rely on imitation learning (IL) due to its better sample efficiency. However, it is costly to collect comprehensive expert demonstrations that enable IL to generalize to all possible scenarios, and any distribution shift would require recollecting data for finetuning. Therefore, RL is appealing if it can build upon IL as an efficient autonomous self-improvement procedure. We propose imitation bootstrapped reinforcement learning (IBRL), a novel framework for sample-efficient RL with demonstrat"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2311.02198","kind":"arxiv","version":6},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2023-11-03T19:03:20Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"de3be4af5970b5ea05b739b0949993ab8d13b78587b0533dd33d338bf0dc7c20","abstract_canon_sha256":"3c1cb246750518f0dad32b05805d07dc474e9f3140978a3c235a34e555799d6c"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:21:09.353982Z","signature_b64":"wgWMRMy/iiTOqb227dBnOhQjR0+paV6U6Y2j7KvBi6xqdNlIMIr/PYbaJUOfJi5f9zfntr0MC6AgL3JYCG7uBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"c1985024735e6cdde1a9cd64a452eccad4abac68093b1e52ffdfcdad3a8a2b2a","last_reissued_at":"2026-07-05T08:21:09.353610Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:21:09.353610Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Imitation Bootstrapped Reinforcement Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Dorsa Sadigh, Hengyuan Hu, Suvir Mirchandani","submitted_at":"2023-11-03T19:03:20Z","abstract_excerpt":"Despite the considerable potential of reinforcement learning (RL), robotic control tasks predominantly rely on imitation learning (IL) due to its better sample efficiency. However, it is costly to collect comprehensive expert demonstrations that enable IL to generalize to all possible scenarios, and any distribution shift would require recollecting data for finetuning. Therefore, RL is appealing if it can build upon IL as an efficient autonomous self-improvement procedure. We propose imitation bootstrapped reinforcement learning (IBRL), a novel framework for sample-efficient RL with demonstrat"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2311.02198","kind":"arxiv","version":6},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2311.02198/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2311.02198","created_at":"2026-07-05T08:21:09.353671+00:00"},{"alias_kind":"arxiv_version","alias_value":"2311.02198v6","created_at":"2026-07-05T08:21:09.353671+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2311.02198","created_at":"2026-07-05T08:21:09.353671+00:00"},{"alias_kind":"pith_short_12","alias_value":"YGMFAJDTLZWN","created_at":"2026-07-05T08:21:09.353671+00:00"},{"alias_kind":"pith_short_16","alias_value":"YGMFAJDTLZWN3YNJ","created_at":"2026-07-05T08:21:09.353671+00:00"},{"alias_kind":"pith_short_8","alias_value":"YGMFAJDT","created_at":"2026-07-05T08:21:09.353671+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":8,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.12372","citing_title":"UniIntervene: Agentic Intervention for Efficient Real-World Reinforcement Learning","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2606.10305","citing_title":"SARM2: Multi-Task Stage Aware Reward Modeling for Self Improving Robotic Manipulation","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2510.04280","citing_title":"A KL-regularization Framework for Learning to Plan with Adaptive Priors","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2509.16615","citing_title":"LLM-Guided Task- and Affordance-Level Exploration in Reinforcement Learning","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2505.18719","citing_title":"VLA-RL: Towards Masterful and General Robotic Manipulation with Scalable Reinforcement Learning","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2409.00588","citing_title":"Diffusion Policy Policy Optimization","ref_index":40,"is_internal_anchor":false},{"citing_arxiv_id":"2602.09580","citing_title":"SERNF: Sample-Efficient Real-World Dexterous Policy Fine-Tuning via Action-Chunked Critics and Normalizing Flows","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11387","citing_title":"Behavioral Mode Discovery for Fine-tuning Multimodal Generative Policies","ref_index":13,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/YGMFAJDTLZWN3YNJZVSKIUXMZL","json":"https://pith.science/pith/YGMFAJDTLZWN3YNJZVSKIUXMZL.json","graph_json":"https://pith.science/api/pith-number/YGMFAJDTLZWN3YNJZVSKIUXMZL/graph.json","events_json":"https://pith.science/api/pith-number/YGMFAJDTLZWN3YNJZVSKIUXMZL/events.json","paper":"https://pith.science/paper/YGMFAJDT"},"agent_actions":{"view_html":"https://pith.science/pith/YGMFAJDTLZWN3YNJZVSKIUXMZL","download_json":"https://pith.science/pith/YGMFAJDTLZWN3YNJZVSKIUXMZL.json","view_paper":"https://pith.science/paper/YGMFAJDT","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2311.02198&json=true","fetch_graph":"https://pith.science/api/pith-number/YGMFAJDTLZWN3YNJZVSKIUXMZL/graph.json","fetch_events":"https://pith.science/api/pith-number/YGMFAJDTLZWN3YNJZVSKIUXMZL/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/YGMFAJDTLZWN3YNJZVSKIUXMZL/action/timestamp_anchor","attest_storage":"https://pith.science/pith/YGMFAJDTLZWN3YNJZVSKIUXMZL/action/storage_attestation","attest_author":"https://pith.science/pith/YGMFAJDTLZWN3YNJZVSKIUXMZL/action/author_attestation","sign_citation":"https://pith.science/pith/YGMFAJDTLZWN3YNJZVSKIUXMZL/action/citation_signature","submit_replication":"https://pith.science/pith/YGMFAJDTLZWN3YNJZVSKIUXMZL/action/replication_record"}},"created_at":"2026-07-05T08:21:09.353671+00:00","updated_at":"2026-07-05T08:21:09.353671+00:00"}