{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:OHNZ2SPDEY5BCIP47ZJ7JGIW2P","short_pith_number":"pith:OHNZ2SPD","schema_version":"1.0","canonical_sha256":"71db9d49e3263a1121fcfe53f49916d3fd542b7ec5b87d91404b47aa801350f4","source":{"kind":"arxiv","id":"2506.07505","version":1},"attestation_state":"computed","paper":{"title":"Reinforcement Learning via Implicit Imitation Guidance","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Alec M. Lessing, Annie S. Chen, Chelsea Finn, Perry Dong","submitted_at":"2025-06-09T07:32:52Z","abstract_excerpt":"We study the problem of sample efficient reinforcement learning, where prior data such as demonstrations are provided for initialization in lieu of a dense reward signal. A natural approach is to incorporate an imitation learning objective, either as regularization during training or to acquire a reference policy. However, imitation learning objectives can ultimately degrade long-term performance, as it does not directly align with reward maximization. In this work, we propose to use prior data solely for guiding exploration via noise added to the policy, sidestepping the need for explicit beh"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2506.07505","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-06-09T07:32:52Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"508813b7cf722eaa48ccba460b5d028e60c006da479997b1b7ccc395fda51b91","abstract_canon_sha256":"d463ccbcd6c4239526b807b94f1af4ccf362f087539efd43eb050b8839ad05db"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:18:28.473516Z","signature_b64":"Ew/7NNHpN/vNgTvw0MzXvB/dUDEdDR2b4XHCMPPO7YyHP3Plx3mk7pe2quzxhgJJRT9Cf968GC+fappTxiG+AA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"71db9d49e3263a1121fcfe53f49916d3fd542b7ec5b87d91404b47aa801350f4","last_reissued_at":"2026-07-05T11:18:28.473037Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:18:28.473037Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Reinforcement Learning via Implicit Imitation Guidance","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Alec M. Lessing, Annie S. Chen, Chelsea Finn, Perry Dong","submitted_at":"2025-06-09T07:32:52Z","abstract_excerpt":"We study the problem of sample efficient reinforcement learning, where prior data such as demonstrations are provided for initialization in lieu of a dense reward signal. A natural approach is to incorporate an imitation learning objective, either as regularization during training or to acquire a reference policy. However, imitation learning objectives can ultimately degrade long-term performance, as it does not directly align with reward maximization. In this work, we propose to use prior data solely for guiding exploration via noise added to the policy, sidestepping the need for explicit beh"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.07505","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2506.07505/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2506.07505","created_at":"2026-07-05T11:18:28.473088+00:00"},{"alias_kind":"arxiv_version","alias_value":"2506.07505v1","created_at":"2026-07-05T11:18:28.473088+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.07505","created_at":"2026-07-05T11:18:28.473088+00:00"},{"alias_kind":"pith_short_12","alias_value":"OHNZ2SPDEY5B","created_at":"2026-07-05T11:18:28.473088+00:00"},{"alias_kind":"pith_short_16","alias_value":"OHNZ2SPDEY5BCIP4","created_at":"2026-07-05T11:18:28.473088+00:00"},{"alias_kind":"pith_short_8","alias_value":"OHNZ2SPD","created_at":"2026-07-05T11:18:28.473088+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2507.07986","citing_title":"EXPO: Stable Reinforcement Learning with Expressive Policies","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2603.13842","citing_title":"Fine-tuning is Not Enough: A Parallel Framework for Collaborative Imitation and Reinforcement Learning in End-to-end Autonomous Driving","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2604.07945","citing_title":"Incremental Residual Reinforcement Learning Toward Real-World Learning for Social Navigation","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2604.19730","citing_title":"FASTER: Value-Guided Sampling for Fast RL","ref_index":41,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/OHNZ2SPDEY5BCIP47ZJ7JGIW2P","json":"https://pith.science/pith/OHNZ2SPDEY5BCIP47ZJ7JGIW2P.json","graph_json":"https://pith.science/api/pith-number/OHNZ2SPDEY5BCIP47ZJ7JGIW2P/graph.json","events_json":"https://pith.science/api/pith-number/OHNZ2SPDEY5BCIP47ZJ7JGIW2P/events.json","paper":"https://pith.science/paper/OHNZ2SPD"},"agent_actions":{"view_html":"https://pith.science/pith/OHNZ2SPDEY5BCIP47ZJ7JGIW2P","download_json":"https://pith.science/pith/OHNZ2SPDEY5BCIP47ZJ7JGIW2P.json","view_paper":"https://pith.science/paper/OHNZ2SPD","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2506.07505&json=true","fetch_graph":"https://pith.science/api/pith-number/OHNZ2SPDEY5BCIP47ZJ7JGIW2P/graph.json","fetch_events":"https://pith.science/api/pith-number/OHNZ2SPDEY5BCIP47ZJ7JGIW2P/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/OHNZ2SPDEY5BCIP47ZJ7JGIW2P/action/timestamp_anchor","attest_storage":"https://pith.science/pith/OHNZ2SPDEY5BCIP47ZJ7JGIW2P/action/storage_attestation","attest_author":"https://pith.science/pith/OHNZ2SPDEY5BCIP47ZJ7JGIW2P/action/author_attestation","sign_citation":"https://pith.science/pith/OHNZ2SPDEY5BCIP47ZJ7JGIW2P/action/citation_signature","submit_replication":"https://pith.science/pith/OHNZ2SPDEY5BCIP47ZJ7JGIW2P/action/replication_record"}},"created_at":"2026-07-05T11:18:28.473088+00:00","updated_at":"2026-07-05T11:18:28.473088+00:00"}