{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:FNDHUI43XZDJJ7IWUDKP7EYLXX","short_pith_number":"pith:FNDHUI43","schema_version":"1.0","canonical_sha256":"2b467a239bbe4694fd16a0d4ff930bbddd49c75c98098c0f440213951dbd4aa7","source":{"kind":"arxiv","id":"2209.11082","version":2},"attestation_state":"computed","paper":{"title":"Bypassing the Simulation-to-reality Gap: Online Reinforcement Learning using a Supervisor","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.RO","authors_text":"Benjamin David Evans, Hendrik W. Jordaan, Herman A. Engelbrecht, Hongrui Zheng, Johannes Betz, Rahul Mangharam","submitted_at":"2022-09-22T15:13:43Z","abstract_excerpt":"Deep reinforcement learning (DRL) is a promising method to learn control policies for robots only from demonstration and experience. To cover the whole dynamic behaviour of the robot, DRL training is an active exploration process typically performed in simulation environments. Although this simulation training is cheap and fast, applying DRL algorithms to real-world settings is difficult. If agents are trained until they perform safely in simulation, transferring them to physical systems is difficult due to the sim-to-real gap caused by the difference between the simulation dynamics and the ph"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2209.11082","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.RO","submitted_at":"2022-09-22T15:13:43Z","cross_cats_sorted":[],"title_canon_sha256":"15993a51ac736cc2d3da75d776a911eead49b9e97d35501f157a8a9aa27b172f","abstract_canon_sha256":"637e4998f87bd0fa578bc1d1bd7352302564cfbc49308ad1ee633efbfec19b3f"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:30:26.052759Z","signature_b64":"TMuKGXHSKf1bJXgDtkxkYDmPGrhNufv1GZh8Qa8U1wqrd3Kypg0VZgtQzo47wCAnQgJAdxCVTN9L/CL2g+FpBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"2b467a239bbe4694fd16a0d4ff930bbddd49c75c98098c0f440213951dbd4aa7","last_reissued_at":"2026-07-05T06:30:26.052218Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:30:26.052218Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Bypassing the Simulation-to-reality Gap: Online Reinforcement Learning using a Supervisor","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.RO","authors_text":"Benjamin David Evans, Hendrik W. Jordaan, Herman A. Engelbrecht, Hongrui Zheng, Johannes Betz, Rahul Mangharam","submitted_at":"2022-09-22T15:13:43Z","abstract_excerpt":"Deep reinforcement learning (DRL) is a promising method to learn control policies for robots only from demonstration and experience. To cover the whole dynamic behaviour of the robot, DRL training is an active exploration process typically performed in simulation environments. Although this simulation training is cheap and fast, applying DRL algorithms to real-world settings is difficult. If agents are trained until they perform safely in simulation, transferring them to physical systems is difficult due to the sim-to-real gap caused by the difference between the simulation dynamics and the ph"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2209.11082","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2209.11082/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2209.11082","created_at":"2026-07-05T06:30:26.052277+00:00"},{"alias_kind":"arxiv_version","alias_value":"2209.11082v2","created_at":"2026-07-05T06:30:26.052277+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2209.11082","created_at":"2026-07-05T06:30:26.052277+00:00"},{"alias_kind":"pith_short_12","alias_value":"FNDHUI43XZDJ","created_at":"2026-07-05T06:30:26.052277+00:00"},{"alias_kind":"pith_short_16","alias_value":"FNDHUI43XZDJJ7IW","created_at":"2026-07-05T06:30:26.052277+00:00"},{"alias_kind":"pith_short_8","alias_value":"FNDHUI43","created_at":"2026-07-05T06:30:26.052277+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2604.09499","citing_title":"Physics-Informed Reinforcement Learning of Spatial Density Velocity Potentials for Map-Free Racing","ref_index":8,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/FNDHUI43XZDJJ7IWUDKP7EYLXX","json":"https://pith.science/pith/FNDHUI43XZDJJ7IWUDKP7EYLXX.json","graph_json":"https://pith.science/api/pith-number/FNDHUI43XZDJJ7IWUDKP7EYLXX/graph.json","events_json":"https://pith.science/api/pith-number/FNDHUI43XZDJJ7IWUDKP7EYLXX/events.json","paper":"https://pith.science/paper/FNDHUI43"},"agent_actions":{"view_html":"https://pith.science/pith/FNDHUI43XZDJJ7IWUDKP7EYLXX","download_json":"https://pith.science/pith/FNDHUI43XZDJJ7IWUDKP7EYLXX.json","view_paper":"https://pith.science/paper/FNDHUI43","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2209.11082&json=true","fetch_graph":"https://pith.science/api/pith-number/FNDHUI43XZDJJ7IWUDKP7EYLXX/graph.json","fetch_events":"https://pith.science/api/pith-number/FNDHUI43XZDJJ7IWUDKP7EYLXX/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/FNDHUI43XZDJJ7IWUDKP7EYLXX/action/timestamp_anchor","attest_storage":"https://pith.science/pith/FNDHUI43XZDJJ7IWUDKP7EYLXX/action/storage_attestation","attest_author":"https://pith.science/pith/FNDHUI43XZDJJ7IWUDKP7EYLXX/action/author_attestation","sign_citation":"https://pith.science/pith/FNDHUI43XZDJJ7IWUDKP7EYLXX/action/citation_signature","submit_replication":"https://pith.science/pith/FNDHUI43XZDJJ7IWUDKP7EYLXX/action/replication_record"}},"created_at":"2026-07-05T06:30:26.052277+00:00","updated_at":"2026-07-05T06:30:26.052277+00:00"}