{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:OLJEP7PRG2HXIVRQHDHGMP2AOF","short_pith_number":"pith:OLJEP7PR","schema_version":"1.0","canonical_sha256":"72d247fdf1368f74563038ce663f40715f030cd670c7f747c99116206fa42202","source":{"kind":"arxiv","id":"2311.12996","version":2},"attestation_state":"computed","paper":{"title":"RLIF: Interactive Imitation Learning as Reinforcement Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.RO"],"primary_cat":"cs.AI","authors_text":"Jianlan Luo, Perry Dong, Sergey Levine, Yi Ma, Yuexiang Zhai","submitted_at":"2023-11-21T21:05:21Z","abstract_excerpt":"Although reinforcement learning methods offer a powerful framework for automatic skill acquisition, for practical learning-based control problems in domains such as robotics, imitation learning often provides a more convenient and accessible alternative. In particular, an interactive imitation learning method such as DAgger, which queries a near-optimal expert to intervene online to collect correction data for addressing the distributional shift challenges that afflict na\\\"ive behavioral cloning, can enjoy good performance both in theory and practice without requiring manually specified reward"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2311.12996","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2023-11-21T21:05:21Z","cross_cats_sorted":["cs.RO"],"title_canon_sha256":"fa7fc3b019d5d8e75e7486c55e0e77f294647e2b7461ecf7566990d78f00df61","abstract_canon_sha256":"6c077e0dbd20b8c2997648850ba9e56bcb68ec952a978e4d5926846944557fac"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:57:46.802888Z","signature_b64":"Yh2os244C3VAWpO9NUoJoDK3vPPikC/HUBmQNH0DI70HPuampnoz5J/9X5R0RXReW/wbkTXcHvcu8pXnLM01Cg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"72d247fdf1368f74563038ce663f40715f030cd670c7f747c99116206fa42202","last_reissued_at":"2026-07-05T07:57:46.802255Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:57:46.802255Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"RLIF: Interactive Imitation Learning as Reinforcement Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.RO"],"primary_cat":"cs.AI","authors_text":"Jianlan Luo, Perry Dong, Sergey Levine, Yi Ma, Yuexiang Zhai","submitted_at":"2023-11-21T21:05:21Z","abstract_excerpt":"Although reinforcement learning methods offer a powerful framework for automatic skill acquisition, for practical learning-based control problems in domains such as robotics, imitation learning often provides a more convenient and accessible alternative. In particular, an interactive imitation learning method such as DAgger, which queries a near-optimal expert to intervene online to collect correction data for addressing the distributional shift challenges that afflict na\\\"ive behavioral cloning, can enjoy good performance both in theory and practice without requiring manually specified reward"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2311.12996","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2311.12996/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2311.12996","created_at":"2026-07-05T07:57:46.802320+00:00"},{"alias_kind":"arxiv_version","alias_value":"2311.12996v2","created_at":"2026-07-05T07:57:46.802320+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2311.12996","created_at":"2026-07-05T07:57:46.802320+00:00"},{"alias_kind":"pith_short_12","alias_value":"OLJEP7PRG2HX","created_at":"2026-07-05T07:57:46.802320+00:00"},{"alias_kind":"pith_short_16","alias_value":"OLJEP7PRG2HXIVRQ","created_at":"2026-07-05T07:57:46.802320+00:00"},{"alias_kind":"pith_short_8","alias_value":"OLJEP7PR","created_at":"2026-07-05T07:57:46.802320+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.10927","citing_title":"AllDayNav: Lifelong Navigation via Real-World Reinforcement Learning","ref_index":61,"is_internal_anchor":false},{"citing_arxiv_id":"2602.09023","citing_title":"TwinRL: Digital Twin-Driven Reinforcement Learning for Real-World Robotic Manipulation","ref_index":40,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/OLJEP7PRG2HXIVRQHDHGMP2AOF","json":"https://pith.science/pith/OLJEP7PRG2HXIVRQHDHGMP2AOF.json","graph_json":"https://pith.science/api/pith-number/OLJEP7PRG2HXIVRQHDHGMP2AOF/graph.json","events_json":"https://pith.science/api/pith-number/OLJEP7PRG2HXIVRQHDHGMP2AOF/events.json","paper":"https://pith.science/paper/OLJEP7PR"},"agent_actions":{"view_html":"https://pith.science/pith/OLJEP7PRG2HXIVRQHDHGMP2AOF","download_json":"https://pith.science/pith/OLJEP7PRG2HXIVRQHDHGMP2AOF.json","view_paper":"https://pith.science/paper/OLJEP7PR","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2311.12996&json=true","fetch_graph":"https://pith.science/api/pith-number/OLJEP7PRG2HXIVRQHDHGMP2AOF/graph.json","fetch_events":"https://pith.science/api/pith-number/OLJEP7PRG2HXIVRQHDHGMP2AOF/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/OLJEP7PRG2HXIVRQHDHGMP2AOF/action/timestamp_anchor","attest_storage":"https://pith.science/pith/OLJEP7PRG2HXIVRQHDHGMP2AOF/action/storage_attestation","attest_author":"https://pith.science/pith/OLJEP7PRG2HXIVRQHDHGMP2AOF/action/author_attestation","sign_citation":"https://pith.science/pith/OLJEP7PRG2HXIVRQHDHGMP2AOF/action/citation_signature","submit_replication":"https://pith.science/pith/OLJEP7PRG2HXIVRQHDHGMP2AOF/action/replication_record"}},"created_at":"2026-07-05T07:57:46.802320+00:00","updated_at":"2026-07-05T07:57:46.802320+00:00"}