{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:MH3XM254YWE5CAK2UNZMBDVTOS","short_pith_number":"pith:MH3XM254","schema_version":"1.0","canonical_sha256":"61f7766bbcc589d1015aa372c08eb37495c02684ded6da1a8c2074806035ee08","source":{"kind":"arxiv","id":"2202.10341","version":1},"attestation_state":"computed","paper":{"title":"Efficient Learning of Safe Driving Policy via Human-AI Copilot Optimization","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.RO"],"primary_cat":"cs.LG","authors_text":"Bolei Zhou, Quanyi Li, Zhenghao Peng","submitted_at":"2022-02-17T06:29:46Z","abstract_excerpt":"Human intervention is an effective way to inject human knowledge into the training loop of reinforcement learning, which can bring fast learning and ensured training safety. Given the very limited budget of human intervention, it remains challenging to design when and how human expert interacts with the learning agent in the training. In this work, we develop a novel human-in-the-loop learning method called Human-AI Copilot Optimization (HACO).To allow the agent's sufficient exploration in the risky environments while ensuring the training safety, the human expert can take over the control and"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2202.10341","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2022-02-17T06:29:46Z","cross_cats_sorted":["cs.AI","cs.RO"],"title_canon_sha256":"17ec48b715c12f21dff8fd2e182ca205d4f605a2aa1c3b4f3d6250b4c6c222d6","abstract_canon_sha256":"aca419607b89179d18aa212f09b8c9df0fab284eed21e70c2f0dd3f9dd27a312"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T03:58:41.476658Z","signature_b64":"9wOp3JSOP8tLoFa6DMG7gq3Nx7HhDsliIN1goW1Wx8eRm7R3pdJrUEfEp6OUmiwRPWJcVxQrOUCnud3TB9N7Bg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"61f7766bbcc589d1015aa372c08eb37495c02684ded6da1a8c2074806035ee08","last_reissued_at":"2026-07-05T03:58:41.476232Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T03:58:41.476232Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Efficient Learning of Safe Driving Policy via Human-AI Copilot Optimization","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.RO"],"primary_cat":"cs.LG","authors_text":"Bolei Zhou, Quanyi Li, Zhenghao Peng","submitted_at":"2022-02-17T06:29:46Z","abstract_excerpt":"Human intervention is an effective way to inject human knowledge into the training loop of reinforcement learning, which can bring fast learning and ensured training safety. Given the very limited budget of human intervention, it remains challenging to design when and how human expert interacts with the learning agent in the training. In this work, we develop a novel human-in-the-loop learning method called Human-AI Copilot Optimization (HACO).To allow the agent's sufficient exploration in the risky environments while ensuring the training safety, the human expert can take over the control and"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2202.10341","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2202.10341/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2202.10341","created_at":"2026-07-05T03:58:41.476287+00:00"},{"alias_kind":"arxiv_version","alias_value":"2202.10341v1","created_at":"2026-07-05T03:58:41.476287+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2202.10341","created_at":"2026-07-05T03:58:41.476287+00:00"},{"alias_kind":"pith_short_12","alias_value":"MH3XM254YWE5","created_at":"2026-07-05T03:58:41.476287+00:00"},{"alias_kind":"pith_short_16","alias_value":"MH3XM254YWE5CAK2","created_at":"2026-07-05T03:58:41.476287+00:00"},{"alias_kind":"pith_short_8","alias_value":"MH3XM254","created_at":"2026-07-05T03:58:41.476287+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.31286","citing_title":"DeMaVLA: A Vision-Language-Action Foundation Model for Generalizable Deformable Manipulation","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2606.28320","citing_title":"WARP-RM: A Warp-Augmented Relative Progress Reward Model for Data Curation","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2605.15971","citing_title":"OHP-RL: Online Human Preference as Guidance in Reinforcement Learning for Robot Manipulation","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2605.15971","citing_title":"OHP-RL: Online Human Preference as Guidance in Reinforcement Learning for Robot Manipulation","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2604.14723","citing_title":"Bounded Autonomy for Enterprise AI: Typed Action Contracts and Consumer-Side Execution","ref_index":23,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/MH3XM254YWE5CAK2UNZMBDVTOS","json":"https://pith.science/pith/MH3XM254YWE5CAK2UNZMBDVTOS.json","graph_json":"https://pith.science/api/pith-number/MH3XM254YWE5CAK2UNZMBDVTOS/graph.json","events_json":"https://pith.science/api/pith-number/MH3XM254YWE5CAK2UNZMBDVTOS/events.json","paper":"https://pith.science/paper/MH3XM254"},"agent_actions":{"view_html":"https://pith.science/pith/MH3XM254YWE5CAK2UNZMBDVTOS","download_json":"https://pith.science/pith/MH3XM254YWE5CAK2UNZMBDVTOS.json","view_paper":"https://pith.science/paper/MH3XM254","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2202.10341&json=true","fetch_graph":"https://pith.science/api/pith-number/MH3XM254YWE5CAK2UNZMBDVTOS/graph.json","fetch_events":"https://pith.science/api/pith-number/MH3XM254YWE5CAK2UNZMBDVTOS/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/MH3XM254YWE5CAK2UNZMBDVTOS/action/timestamp_anchor","attest_storage":"https://pith.science/pith/MH3XM254YWE5CAK2UNZMBDVTOS/action/storage_attestation","attest_author":"https://pith.science/pith/MH3XM254YWE5CAK2UNZMBDVTOS/action/author_attestation","sign_citation":"https://pith.science/pith/MH3XM254YWE5CAK2UNZMBDVTOS/action/citation_signature","submit_replication":"https://pith.science/pith/MH3XM254YWE5CAK2UNZMBDVTOS/action/replication_record"}},"created_at":"2026-07-05T03:58:41.476287+00:00","updated_at":"2026-07-05T03:58:41.476287+00:00"}