{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:J6HRIENBQYYSE3D26N5TUJPBLT","short_pith_number":"pith:J6HRIENB","schema_version":"1.0","canonical_sha256":"4f8f1411a18631226c7af37b3a25e15ce99b17f1bdb9efb6badbae427066242b","source":{"kind":"arxiv","id":"2203.12759","version":3},"attestation_state":"computed","paper":{"title":"Asynchronous Reinforcement Learning for Real-Time Control of Physical Robots","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.RO","authors_text":"A. Rupam Mahmood, Yufeng Yuan","submitted_at":"2022-03-23T23:05:28Z","abstract_excerpt":"An oft-ignored challenge of real-world reinforcement learning is that the real world does not pause when agents make learning updates. As standard simulated environments do not address this real-time aspect of learning, most available implementations of RL algorithms process environment interactions and learning updates sequentially. As a consequence, when such implementations are deployed in the real world, they may make decisions based on significantly delayed observations and not act responsively. Asynchronous learning has been proposed to solve this issue, but no systematic comparison betw"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2203.12759","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.RO","submitted_at":"2022-03-23T23:05:28Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"04947d4da71e352b7bb2ec550f74e9d03d30ad2cb27e703f2347242006a28d6f","abstract_canon_sha256":"8ef94a01ba1b08b554e4ff3583bcb394fde1a1ebd60e15f8f38f3bca55dd972f"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T04:10:26.781640Z","signature_b64":"kbMqRQBc6J8FlaGENsMW3EJAoWGzpTdSAry+bggTGUZhP1hTiIrbpgMRFpJc5KBQj97kIOeRpZNphJkY5VHCCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"4f8f1411a18631226c7af37b3a25e15ce99b17f1bdb9efb6badbae427066242b","last_reissued_at":"2026-07-05T04:10:26.781077Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T04:10:26.781077Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Asynchronous Reinforcement Learning for Real-Time Control of Physical Robots","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.RO","authors_text":"A. Rupam Mahmood, Yufeng Yuan","submitted_at":"2022-03-23T23:05:28Z","abstract_excerpt":"An oft-ignored challenge of real-world reinforcement learning is that the real world does not pause when agents make learning updates. As standard simulated environments do not address this real-time aspect of learning, most available implementations of RL algorithms process environment interactions and learning updates sequentially. As a consequence, when such implementations are deployed in the real world, they may make decisions based on significantly delayed observations and not act responsively. Asynchronous learning has been proposed to solve this issue, but no systematic comparison betw"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2203.12759","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2203.12759/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2203.12759","created_at":"2026-07-05T04:10:26.781137+00:00"},{"alias_kind":"arxiv_version","alias_value":"2203.12759v3","created_at":"2026-07-05T04:10:26.781137+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2203.12759","created_at":"2026-07-05T04:10:26.781137+00:00"},{"alias_kind":"pith_short_12","alias_value":"J6HRIENBQYYS","created_at":"2026-07-05T04:10:26.781137+00:00"},{"alias_kind":"pith_short_16","alias_value":"J6HRIENBQYYSE3D2","created_at":"2026-07-05T04:10:26.781137+00:00"},{"alias_kind":"pith_short_8","alias_value":"J6HRIENB","created_at":"2026-07-05T04:10:26.781137+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2507.10814","citing_title":"Versatile and Generalizable Manipulation via Goal-Conditioned Reinforcement Learning with Grounded Object Detection","ref_index":11,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/J6HRIENBQYYSE3D26N5TUJPBLT","json":"https://pith.science/pith/J6HRIENBQYYSE3D26N5TUJPBLT.json","graph_json":"https://pith.science/api/pith-number/J6HRIENBQYYSE3D26N5TUJPBLT/graph.json","events_json":"https://pith.science/api/pith-number/J6HRIENBQYYSE3D26N5TUJPBLT/events.json","paper":"https://pith.science/paper/J6HRIENB"},"agent_actions":{"view_html":"https://pith.science/pith/J6HRIENBQYYSE3D26N5TUJPBLT","download_json":"https://pith.science/pith/J6HRIENBQYYSE3D26N5TUJPBLT.json","view_paper":"https://pith.science/paper/J6HRIENB","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2203.12759&json=true","fetch_graph":"https://pith.science/api/pith-number/J6HRIENBQYYSE3D26N5TUJPBLT/graph.json","fetch_events":"https://pith.science/api/pith-number/J6HRIENBQYYSE3D26N5TUJPBLT/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/J6HRIENBQYYSE3D26N5TUJPBLT/action/timestamp_anchor","attest_storage":"https://pith.science/pith/J6HRIENBQYYSE3D26N5TUJPBLT/action/storage_attestation","attest_author":"https://pith.science/pith/J6HRIENBQYYSE3D26N5TUJPBLT/action/author_attestation","sign_citation":"https://pith.science/pith/J6HRIENBQYYSE3D26N5TUJPBLT/action/citation_signature","submit_replication":"https://pith.science/pith/J6HRIENBQYYSE3D26N5TUJPBLT/action/replication_record"}},"created_at":"2026-07-05T04:10:26.781137+00:00","updated_at":"2026-07-05T04:10:26.781137+00:00"}