{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:ZSOPWTXTXJB3GW3YQWNFVP26QU","short_pith_number":"pith:ZSOPWTXT","schema_version":"1.0","canonical_sha256":"cc9cfb4ef3ba43b35b78859a5abf5e852cd1fa784ddef417409e4b86aa5ff612","source":{"kind":"arxiv","id":"2301.00051","version":2},"attestation_state":"computed","paper":{"title":"Learning from Guided Play: Improving Exploration for Adversarial Imitation Learning with Simple Auxiliary Tasks","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.RO"],"primary_cat":"cs.LG","authors_text":"Bryan Chan, Jonathan Kelly, Trevor Ablett","submitted_at":"2022-12-30T20:38:54Z","abstract_excerpt":"Adversarial imitation learning (AIL) has become a popular alternative to supervised imitation learning that reduces the distribution shift suffered by the latter. However, AIL requires effective exploration during an online reinforcement learning phase. In this work, we show that the standard, naive approach to exploration can manifest as a suboptimal local maximum if a policy learned with AIL sufficiently matches the expert distribution without fully learning the desired task. This can be particularly catastrophic for manipulation tasks, where the difference between an expert and a non-expert"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2301.00051","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2022-12-30T20:38:54Z","cross_cats_sorted":["cs.AI","cs.RO"],"title_canon_sha256":"d15204378419cf192612952695a1f8be4582e43b6b7e65aace8bdebd393ee357","abstract_canon_sha256":"22c835515a2c827aa0c6261d6a6ac65e3b619d9d670d5359c1b20bb3869ae165"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:00:20.602597Z","signature_b64":"PCP71TtQvTY6B/J7QNsIgoE1uKYwef7t8BcrUy1Yv1DYkHm4Jk8aApmVGfHG1dgz5zbT87c9Ngf25OqQz7exCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"cc9cfb4ef3ba43b35b78859a5abf5e852cd1fa784ddef417409e4b86aa5ff612","last_reissued_at":"2026-07-05T07:00:20.602133Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:00:20.602133Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Learning from Guided Play: Improving Exploration for Adversarial Imitation Learning with Simple Auxiliary Tasks","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.RO"],"primary_cat":"cs.LG","authors_text":"Bryan Chan, Jonathan Kelly, Trevor Ablett","submitted_at":"2022-12-30T20:38:54Z","abstract_excerpt":"Adversarial imitation learning (AIL) has become a popular alternative to supervised imitation learning that reduces the distribution shift suffered by the latter. However, AIL requires effective exploration during an online reinforcement learning phase. In this work, we show that the standard, naive approach to exploration can manifest as a suboptimal local maximum if a policy learned with AIL sufficiently matches the expert distribution without fully learning the desired task. This can be particularly catastrophic for manipulation tasks, where the difference between an expert and a non-expert"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2301.00051","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2301.00051/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2301.00051","created_at":"2026-07-05T07:00:20.602201+00:00"},{"alias_kind":"arxiv_version","alias_value":"2301.00051v2","created_at":"2026-07-05T07:00:20.602201+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2301.00051","created_at":"2026-07-05T07:00:20.602201+00:00"},{"alias_kind":"pith_short_12","alias_value":"ZSOPWTXTXJB3","created_at":"2026-07-05T07:00:20.602201+00:00"},{"alias_kind":"pith_short_16","alias_value":"ZSOPWTXTXJB3GW3Y","created_at":"2026-07-05T07:00:20.602201+00:00"},{"alias_kind":"pith_short_8","alias_value":"ZSOPWTXT","created_at":"2026-07-05T07:00:20.602201+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ZSOPWTXTXJB3GW3YQWNFVP26QU","json":"https://pith.science/pith/ZSOPWTXTXJB3GW3YQWNFVP26QU.json","graph_json":"https://pith.science/api/pith-number/ZSOPWTXTXJB3GW3YQWNFVP26QU/graph.json","events_json":"https://pith.science/api/pith-number/ZSOPWTXTXJB3GW3YQWNFVP26QU/events.json","paper":"https://pith.science/paper/ZSOPWTXT"},"agent_actions":{"view_html":"https://pith.science/pith/ZSOPWTXTXJB3GW3YQWNFVP26QU","download_json":"https://pith.science/pith/ZSOPWTXTXJB3GW3YQWNFVP26QU.json","view_paper":"https://pith.science/paper/ZSOPWTXT","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2301.00051&json=true","fetch_graph":"https://pith.science/api/pith-number/ZSOPWTXTXJB3GW3YQWNFVP26QU/graph.json","fetch_events":"https://pith.science/api/pith-number/ZSOPWTXTXJB3GW3YQWNFVP26QU/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ZSOPWTXTXJB3GW3YQWNFVP26QU/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ZSOPWTXTXJB3GW3YQWNFVP26QU/action/storage_attestation","attest_author":"https://pith.science/pith/ZSOPWTXTXJB3GW3YQWNFVP26QU/action/author_attestation","sign_citation":"https://pith.science/pith/ZSOPWTXTXJB3GW3YQWNFVP26QU/action/citation_signature","submit_replication":"https://pith.science/pith/ZSOPWTXTXJB3GW3YQWNFVP26QU/action/replication_record"}},"created_at":"2026-07-05T07:00:20.602201+00:00","updated_at":"2026-07-05T07:00:20.602201+00:00"}