{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2021:NB7PUXD66FYTOXCDBYSNUOWFX4","short_pith_number":"pith:NB7PUXD6","canonical_record":{"source":{"id":"2107.08183","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2021-07-17T05:02:25Z","cross_cats_sorted":[],"title_canon_sha256":"bd64e03a0fd7a0d84635cb18af337fa769218b11455aa5cdd42eddc463647716","abstract_canon_sha256":"37bb28c691e9fec18514b8d13dd3436eec30ec50ec07ef88ecda1f12ab7ff668"},"schema_version":"1.0"},"canonical_sha256":"687efa5c7ef171375c430e24da3ac5bf14e3767c8aab8ffae9255d89d2527399","source":{"kind":"arxiv","id":"2107.08183","version":1},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2107.08183","created_at":"2026-07-05T02:58:46Z"},{"alias_kind":"arxiv_version","alias_value":"2107.08183v1","created_at":"2026-07-05T02:58:46Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2107.08183","created_at":"2026-07-05T02:58:46Z"},{"alias_kind":"pith_short_12","alias_value":"NB7PUXD66FYT","created_at":"2026-07-05T02:58:46Z"},{"alias_kind":"pith_short_16","alias_value":"NB7PUXD66FYTOXCD","created_at":"2026-07-05T02:58:46Z"},{"alias_kind":"pith_short_8","alias_value":"NB7PUXD6","created_at":"2026-07-05T02:58:46Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2021:NB7PUXD66FYTOXCDBYSNUOWFX4","target":"record","payload":{"canonical_record":{"source":{"id":"2107.08183","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2021-07-17T05:02:25Z","cross_cats_sorted":[],"title_canon_sha256":"bd64e03a0fd7a0d84635cb18af337fa769218b11455aa5cdd42eddc463647716","abstract_canon_sha256":"37bb28c691e9fec18514b8d13dd3436eec30ec50ec07ef88ecda1f12ab7ff668"},"schema_version":"1.0"},"canonical_sha256":"687efa5c7ef171375c430e24da3ac5bf14e3767c8aab8ffae9255d89d2527399","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T02:58:46.329420Z","signature_b64":"fLPYpB+roWOmMhhSmnKOVJi7fgwkjPAvWFHSur6srxa4860Q1eSCRBiONjPc54i8ePce8gL8zU39rVUmycTCCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"687efa5c7ef171375c430e24da3ac5bf14e3767c8aab8ffae9255d89d2527399","last_reissued_at":"2026-07-05T02:58:46.328965Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T02:58:46.328965Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2107.08183","source_version":1,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T02:58:46Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"jE/MHY4pHwPCcja84GDiybAGDhFUaqbzZaSOg+0f2uZVjr9ImcONK5nyZmCE7TrMAdUDHkNnM1tkFqtBOTAzCw==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-16T09:26:46.033284Z"},"content_sha256":"54577a4f86e5317cd0024f08aed25183add993c99d335e9791186b58ad4fe5be","schema_version":"1.0","event_id":"sha256:54577a4f86e5317cd0024f08aed25183add993c99d335e9791186b58ad4fe5be"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2021:NB7PUXD66FYTOXCDBYSNUOWFX4","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Hierarchical Reinforcement Learning with Optimal Level Synchronization based on a Deep Generative Model","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Christy Liang, Farookh Hussain, Jaeyoon Kim, Junyu Xuan","submitted_at":"2021-07-17T05:02:25Z","abstract_excerpt":"The high-dimensional or sparse reward task of a reinforcement learning (RL) environment requires a superior potential controller such as hierarchical reinforcement learning (HRL) rather than an atomic RL because it absorbs the complexity of commands to achieve the purpose of the task in its hierarchical structure. One of the HRL issues is how to train each level policy with the optimal data collection from its experience. That is to say, how to synchronize adjacent level policies optimally. Our research finds that a HRL model through the off-policy correction technique of HRL, which trains a h"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2107.08183","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2107.08183/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T02:58:46Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"j9c+qaklwyGM+9+TEw/zB1xYJTbMYid31+/I0M+3e/xmd93RlgGpUleSpZgJzDIdmhrgUcR/kfVklHRzNfpnDw==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-16T09:26:46.033790Z"},"content_sha256":"edb097b490559f519770dd6213eb2a1d11c40a83db3b4211719e506412bad9fb","schema_version":"1.0","event_id":"sha256:edb097b490559f519770dd6213eb2a1d11c40a83db3b4211719e506412bad9fb"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/NB7PUXD66FYTOXCDBYSNUOWFX4/bundle.json","state_url":"https://pith.science/pith/NB7PUXD66FYTOXCDBYSNUOWFX4/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/NB7PUXD66FYTOXCDBYSNUOWFX4/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-16T09:26:46Z","links":{"resolver":"https://pith.science/pith/NB7PUXD66FYTOXCDBYSNUOWFX4","bundle":"https://pith.science/pith/NB7PUXD66FYTOXCDBYSNUOWFX4/bundle.json","state":"https://pith.science/pith/NB7PUXD66FYTOXCDBYSNUOWFX4/state.json","well_known_bundle":"https://pith.science/.well-known/pith/NB7PUXD66FYTOXCDBYSNUOWFX4/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2021:NB7PUXD66FYTOXCDBYSNUOWFX4","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"37bb28c691e9fec18514b8d13dd3436eec30ec50ec07ef88ecda1f12ab7ff668","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2021-07-17T05:02:25Z","title_canon_sha256":"bd64e03a0fd7a0d84635cb18af337fa769218b11455aa5cdd42eddc463647716"},"schema_version":"1.0","source":{"id":"2107.08183","kind":"arxiv","version":1}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2107.08183","created_at":"2026-07-05T02:58:46Z"},{"alias_kind":"arxiv_version","alias_value":"2107.08183v1","created_at":"2026-07-05T02:58:46Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2107.08183","created_at":"2026-07-05T02:58:46Z"},{"alias_kind":"pith_short_12","alias_value":"NB7PUXD66FYT","created_at":"2026-07-05T02:58:46Z"},{"alias_kind":"pith_short_16","alias_value":"NB7PUXD66FYTOXCD","created_at":"2026-07-05T02:58:46Z"},{"alias_kind":"pith_short_8","alias_value":"NB7PUXD6","created_at":"2026-07-05T02:58:46Z"}],"graph_snapshots":[{"event_id":"sha256:edb097b490559f519770dd6213eb2a1d11c40a83db3b4211719e506412bad9fb","target":"graph","created_at":"2026-07-05T02:58:46Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2107.08183/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"The high-dimensional or sparse reward task of a reinforcement learning (RL) environment requires a superior potential controller such as hierarchical reinforcement learning (HRL) rather than an atomic RL because it absorbs the complexity of commands to achieve the purpose of the task in its hierarchical structure. One of the HRL issues is how to train each level policy with the optimal data collection from its experience. That is to say, how to synchronize adjacent level policies optimally. Our research finds that a HRL model through the off-policy correction technique of HRL, which trains a h","authors_text":"Christy Liang, Farookh Hussain, Jaeyoon Kim, Junyu Xuan","cross_cats":[],"headline":"","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2021-07-17T05:02:25Z","title":"Hierarchical Reinforcement Learning with Optimal Level Synchronization based on a Deep Generative Model"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2107.08183","kind":"arxiv","version":1},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:54577a4f86e5317cd0024f08aed25183add993c99d335e9791186b58ad4fe5be","target":"record","created_at":"2026-07-05T02:58:46Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"37bb28c691e9fec18514b8d13dd3436eec30ec50ec07ef88ecda1f12ab7ff668","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2021-07-17T05:02:25Z","title_canon_sha256":"bd64e03a0fd7a0d84635cb18af337fa769218b11455aa5cdd42eddc463647716"},"schema_version":"1.0","source":{"id":"2107.08183","kind":"arxiv","version":1}},"canonical_sha256":"687efa5c7ef171375c430e24da3ac5bf14e3767c8aab8ffae9255d89d2527399","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"687efa5c7ef171375c430e24da3ac5bf14e3767c8aab8ffae9255d89d2527399","first_computed_at":"2026-07-05T02:58:46.328965Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T02:58:46.328965Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"fLPYpB+roWOmMhhSmnKOVJi7fgwkjPAvWFHSur6srxa4860Q1eSCRBiONjPc54i8ePce8gL8zU39rVUmycTCCw==","signature_status":"signed_v1","signed_at":"2026-07-05T02:58:46.329420Z","signed_message":"canonical_sha256_bytes"},"source_id":"2107.08183","source_kind":"arxiv","source_version":1}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:54577a4f86e5317cd0024f08aed25183add993c99d335e9791186b58ad4fe5be","sha256:edb097b490559f519770dd6213eb2a1d11c40a83db3b4211719e506412bad9fb"],"state_sha256":"4d1d8fb65ad5a9ca99a300d0976177b9569d38921a5f26d9249d306e9096e93f"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"jW4mn+SARGGZPlT6vBzE+JpOYwkJzYLSrIsORZppBBezk7xeI9ZzABRo3Ds9KJHy9d8hl/bS0a3KGbPog6mfBA==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-16T09:26:46.037490Z","bundle_sha256":"87b9ae2f11765e902ec5aa7decd73040ead8037020e406f69396420d2b8b832e"}}