{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2019:FBZMWRW5OS2PRZBP3BNXXJ6SW5","short_pith_number":"pith:FBZMWRW5","schema_version":"1.0","canonical_sha256":"2872cb46dd74b4f8e42fd85b7ba7d2b77d797e5fff49cd9c5d330d9174b82393","source":{"kind":"arxiv","id":"1912.02241","version":1},"attestation_state":"computed","paper":{"title":"Learning from Interventions using Hierarchical Policies for Safe Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CV","cs.HC","cs.LG"],"primary_cat":"cs.RO","authors_text":"Chenliang Xu, Jing Bi, Tianyou Xiao, Vikas Dhiman","submitted_at":"2019-12-04T20:28:51Z","abstract_excerpt":"Learning from Demonstrations (LfD) via Behavior Cloning (BC) works well on multiple complex tasks. However, a limitation of the typical LfD approach is that it requires expert demonstrations for all scenarios, including those in which the algorithm is already well-trained. The recently proposed Learning from Interventions (LfI) overcomes this limitation by using an expert overseer. The expert overseer only intervenes when it suspects that an unsafe action is about to be taken. Although LfI significantly improves over LfD, the state-of-the-art LfI fails to account for delay caused by the expert"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"1912.02241","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.RO","submitted_at":"2019-12-04T20:28:51Z","cross_cats_sorted":["cs.AI","cs.CV","cs.HC","cs.LG"],"title_canon_sha256":"a842c98c2fb3f17db22a810f8bd358f1da76807d84c959e5ec873c333aacd7d0","abstract_canon_sha256":"2e9e7d73df915cd2fd8949b1092222a40504e318a8fdf30b1495690658b6c4fe"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T00:24:11.676960Z","signature_b64":"3zuylxIkuIOFEoO2cJnNkgrOaVcb0sVxEcZCSutHEkBqKcODUabz40q+1vRYt481ngj/ICZ8CuBODBTGCCCpDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"2872cb46dd74b4f8e42fd85b7ba7d2b77d797e5fff49cd9c5d330d9174b82393","last_reissued_at":"2026-07-05T00:24:11.676498Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T00:24:11.676498Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Learning from Interventions using Hierarchical Policies for Safe Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CV","cs.HC","cs.LG"],"primary_cat":"cs.RO","authors_text":"Chenliang Xu, Jing Bi, Tianyou Xiao, Vikas Dhiman","submitted_at":"2019-12-04T20:28:51Z","abstract_excerpt":"Learning from Demonstrations (LfD) via Behavior Cloning (BC) works well on multiple complex tasks. However, a limitation of the typical LfD approach is that it requires expert demonstrations for all scenarios, including those in which the algorithm is already well-trained. The recently proposed Learning from Interventions (LfI) overcomes this limitation by using an expert overseer. The expert overseer only intervenes when it suspects that an unsafe action is about to be taken. Although LfI significantly improves over LfD, the state-of-the-art LfI fails to account for delay caused by the expert"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1912.02241","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/1912.02241/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"1912.02241","created_at":"2026-07-05T00:24:11.676554+00:00"},{"alias_kind":"arxiv_version","alias_value":"1912.02241v1","created_at":"2026-07-05T00:24:11.676554+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1912.02241","created_at":"2026-07-05T00:24:11.676554+00:00"},{"alias_kind":"pith_short_12","alias_value":"FBZMWRW5OS2P","created_at":"2026-07-05T00:24:11.676554+00:00"},{"alias_kind":"pith_short_16","alias_value":"FBZMWRW5OS2PRZBP","created_at":"2026-07-05T00:24:11.676554+00:00"},{"alias_kind":"pith_short_8","alias_value":"FBZMWRW5","created_at":"2026-07-05T00:24:11.676554+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/FBZMWRW5OS2PRZBP3BNXXJ6SW5","json":"https://pith.science/pith/FBZMWRW5OS2PRZBP3BNXXJ6SW5.json","graph_json":"https://pith.science/api/pith-number/FBZMWRW5OS2PRZBP3BNXXJ6SW5/graph.json","events_json":"https://pith.science/api/pith-number/FBZMWRW5OS2PRZBP3BNXXJ6SW5/events.json","paper":"https://pith.science/paper/FBZMWRW5"},"agent_actions":{"view_html":"https://pith.science/pith/FBZMWRW5OS2PRZBP3BNXXJ6SW5","download_json":"https://pith.science/pith/FBZMWRW5OS2PRZBP3BNXXJ6SW5.json","view_paper":"https://pith.science/paper/FBZMWRW5","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=1912.02241&json=true","fetch_graph":"https://pith.science/api/pith-number/FBZMWRW5OS2PRZBP3BNXXJ6SW5/graph.json","fetch_events":"https://pith.science/api/pith-number/FBZMWRW5OS2PRZBP3BNXXJ6SW5/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/FBZMWRW5OS2PRZBP3BNXXJ6SW5/action/timestamp_anchor","attest_storage":"https://pith.science/pith/FBZMWRW5OS2PRZBP3BNXXJ6SW5/action/storage_attestation","attest_author":"https://pith.science/pith/FBZMWRW5OS2PRZBP3BNXXJ6SW5/action/author_attestation","sign_citation":"https://pith.science/pith/FBZMWRW5OS2PRZBP3BNXXJ6SW5/action/citation_signature","submit_replication":"https://pith.science/pith/FBZMWRW5OS2PRZBP3BNXXJ6SW5/action/replication_record"}},"created_at":"2026-07-05T00:24:11.676554+00:00","updated_at":"2026-07-05T00:24:11.676554+00:00"}