{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2023:P6U5UQI3ZNLQHYTCARVB4KQUDW","short_pith_number":"pith:P6U5UQI3","canonical_record":{"source":{"id":"2312.04704","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.DC","submitted_at":"2023-12-07T21:19:57Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"8c5ea21d80f34dd3a7ceb889019e206cd97c6d34ab507a39a3eb486604767cc1","abstract_canon_sha256":"e4fc9ac9eeb308a15b5e075c4f0a351affb188f4dc6a47edebf060c8be38d3e1"},"schema_version":"1.0"},"canonical_sha256":"7fa9da411bcb5703e262046a1e2a141dad9219bf06a0c658146a052d51b87261","source":{"kind":"arxiv","id":"2312.04704","version":2},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2312.04704","created_at":"2026-07-05T07:41:00Z"},{"alias_kind":"arxiv_version","alias_value":"2312.04704v2","created_at":"2026-07-05T07:41:00Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2312.04704","created_at":"2026-07-05T07:41:00Z"},{"alias_kind":"pith_short_12","alias_value":"P6U5UQI3ZNLQ","created_at":"2026-07-05T07:41:00Z"},{"alias_kind":"pith_short_16","alias_value":"P6U5UQI3ZNLQHYTC","created_at":"2026-07-05T07:41:00Z"},{"alias_kind":"pith_short_8","alias_value":"P6U5UQI3","created_at":"2026-07-05T07:41:00Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2023:P6U5UQI3ZNLQHYTCARVB4KQUDW","target":"record","payload":{"canonical_record":{"source":{"id":"2312.04704","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.DC","submitted_at":"2023-12-07T21:19:57Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"8c5ea21d80f34dd3a7ceb889019e206cd97c6d34ab507a39a3eb486604767cc1","abstract_canon_sha256":"e4fc9ac9eeb308a15b5e075c4f0a351affb188f4dc6a47edebf060c8be38d3e1"},"schema_version":"1.0"},"canonical_sha256":"7fa9da411bcb5703e262046a1e2a141dad9219bf06a0c658146a052d51b87261","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:41:00.882066Z","signature_b64":"KXDyKRUoJ6cSUKuNJiz67x3fdIknUGsEsSRrQ730/vHlKytJbMorco35cvnwRBlleLU+rUSpR8KCeAG3Rm3sCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"7fa9da411bcb5703e262046a1e2a141dad9219bf06a0c658146a052d51b87261","last_reissued_at":"2026-07-05T07:41:00.881545Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:41:00.881545Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2312.04704","source_version":2,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T07:41:00Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"o0vd3TGiEsvDqhtvMy83PsHRJgx6riiH4QVCpyBH8CnJv+1hv958V9awoHnQseEJJ0tm+qNYKIa74RoyUVU8BQ==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-23T21:21:57.083567Z"},"content_sha256":"eeffd1c7df78e4fd785774505a0cdb5528076ecba09f63fb4cdecc02c7795e54","schema_version":"1.0","event_id":"sha256:eeffd1c7df78e4fd785774505a0cdb5528076ecba09f63fb4cdecc02c7795e54"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2023:P6U5UQI3ZNLQHYTCARVB4KQUDW","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Efficient Parallel Reinforcement Learning Framework using the Reactor Model","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.DC","authors_text":"Edward A. Lee, Jacky Kwok, Marten Lohstroh","submitted_at":"2023-12-07T21:19:57Z","abstract_excerpt":"Parallel Reinforcement Learning (RL) frameworks are essential for mapping RL workloads to multiple computational resources, allowing for faster generation of samples, estimation of values, and policy improvement. These computational paradigms require a seamless integration of training, serving, and simulation workloads. Existing frameworks, such as Ray, are not managing this orchestration efficiently, especially in RL tasks that demand intensive input/output and synchronization between actors on a single node. In this study, we have proposed a solution implementing the reactor model, which enf"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2312.04704","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2312.04704/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T07:41:00Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"vfVUrtoAyPy1Pa0R27u5RJWHk+VKVqAHOqUBylkG8cdytnkpaIEmwdXfo4dJikV1YIO6ptqMzQdF/sg7CV6FAw==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-23T21:21:57.084199Z"},"content_sha256":"c3209916fd469b4c7457b87705ae2d1fe6b76919612b1718d4f6eb387324f100","schema_version":"1.0","event_id":"sha256:c3209916fd469b4c7457b87705ae2d1fe6b76919612b1718d4f6eb387324f100"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/P6U5UQI3ZNLQHYTCARVB4KQUDW/bundle.json","state_url":"https://pith.science/pith/P6U5UQI3ZNLQHYTCARVB4KQUDW/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/P6U5UQI3ZNLQHYTCARVB4KQUDW/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-23T21:21:57Z","links":{"resolver":"https://pith.science/pith/P6U5UQI3ZNLQHYTCARVB4KQUDW","bundle":"https://pith.science/pith/P6U5UQI3ZNLQHYTCARVB4KQUDW/bundle.json","state":"https://pith.science/pith/P6U5UQI3ZNLQHYTCARVB4KQUDW/state.json","well_known_bundle":"https://pith.science/.well-known/pith/P6U5UQI3ZNLQHYTCARVB4KQUDW/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2023:P6U5UQI3ZNLQHYTCARVB4KQUDW","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"e4fc9ac9eeb308a15b5e075c4f0a351affb188f4dc6a47edebf060c8be38d3e1","cross_cats_sorted":["cs.LG"],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.DC","submitted_at":"2023-12-07T21:19:57Z","title_canon_sha256":"8c5ea21d80f34dd3a7ceb889019e206cd97c6d34ab507a39a3eb486604767cc1"},"schema_version":"1.0","source":{"id":"2312.04704","kind":"arxiv","version":2}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2312.04704","created_at":"2026-07-05T07:41:00Z"},{"alias_kind":"arxiv_version","alias_value":"2312.04704v2","created_at":"2026-07-05T07:41:00Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2312.04704","created_at":"2026-07-05T07:41:00Z"},{"alias_kind":"pith_short_12","alias_value":"P6U5UQI3ZNLQ","created_at":"2026-07-05T07:41:00Z"},{"alias_kind":"pith_short_16","alias_value":"P6U5UQI3ZNLQHYTC","created_at":"2026-07-05T07:41:00Z"},{"alias_kind":"pith_short_8","alias_value":"P6U5UQI3","created_at":"2026-07-05T07:41:00Z"}],"graph_snapshots":[{"event_id":"sha256:c3209916fd469b4c7457b87705ae2d1fe6b76919612b1718d4f6eb387324f100","target":"graph","created_at":"2026-07-05T07:41:00Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2312.04704/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Parallel Reinforcement Learning (RL) frameworks are essential for mapping RL workloads to multiple computational resources, allowing for faster generation of samples, estimation of values, and policy improvement. These computational paradigms require a seamless integration of training, serving, and simulation workloads. Existing frameworks, such as Ray, are not managing this orchestration efficiently, especially in RL tasks that demand intensive input/output and synchronization between actors on a single node. In this study, we have proposed a solution implementing the reactor model, which enf","authors_text":"Edward A. Lee, Jacky Kwok, Marten Lohstroh","cross_cats":["cs.LG"],"headline":"","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.DC","submitted_at":"2023-12-07T21:19:57Z","title":"Efficient Parallel Reinforcement Learning Framework using the Reactor Model"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2312.04704","kind":"arxiv","version":2},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:eeffd1c7df78e4fd785774505a0cdb5528076ecba09f63fb4cdecc02c7795e54","target":"record","created_at":"2026-07-05T07:41:00Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"e4fc9ac9eeb308a15b5e075c4f0a351affb188f4dc6a47edebf060c8be38d3e1","cross_cats_sorted":["cs.LG"],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.DC","submitted_at":"2023-12-07T21:19:57Z","title_canon_sha256":"8c5ea21d80f34dd3a7ceb889019e206cd97c6d34ab507a39a3eb486604767cc1"},"schema_version":"1.0","source":{"id":"2312.04704","kind":"arxiv","version":2}},"canonical_sha256":"7fa9da411bcb5703e262046a1e2a141dad9219bf06a0c658146a052d51b87261","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"7fa9da411bcb5703e262046a1e2a141dad9219bf06a0c658146a052d51b87261","first_computed_at":"2026-07-05T07:41:00.881545Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T07:41:00.881545Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"KXDyKRUoJ6cSUKuNJiz67x3fdIknUGsEsSRrQ730/vHlKytJbMorco35cvnwRBlleLU+rUSpR8KCeAG3Rm3sCA==","signature_status":"signed_v1","signed_at":"2026-07-05T07:41:00.882066Z","signed_message":"canonical_sha256_bytes"},"source_id":"2312.04704","source_kind":"arxiv","source_version":2}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:eeffd1c7df78e4fd785774505a0cdb5528076ecba09f63fb4cdecc02c7795e54","sha256:c3209916fd469b4c7457b87705ae2d1fe6b76919612b1718d4f6eb387324f100"],"state_sha256":"36023ceeb3208835b2e9f18d21fb1f4a7f5334dd1c08cc0f1ffd75dd84c2846c"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"84hfDy3AefoOkMlo597Uboc00Igl6pwqByr7y05lWT35DpCuUrstNSbDeWPPYDhf/+/FEEKTxczaavSyJzPmCA==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-23T21:21:57.090160Z","bundle_sha256":"3e3495bda708251baf943ca94ff4de5be07dac510fc979216e895705f2a506b7"}}