{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2022:AN5GUGI25DHKUOIJD2TZW6PST5","short_pith_number":"pith:AN5GUGI2","canonical_record":{"source":{"id":"2202.07414","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2022-02-15T14:04:44Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"e1904713a469f2b71e5829a20f2bacd1f7cede564b92cdeb6d671baccc205758","abstract_canon_sha256":"a17eecca836ba059c563933fdaade51d93adb31edaaf3c24ef25bbeca3b56b59"},"schema_version":"1.0"},"canonical_sha256":"037a6a191ae8ceaa39091ea79b79f29f6b3d1bcf54794347ccb895fa2c707da5","source":{"kind":"arxiv","id":"2202.07414","version":1},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2202.07414","created_at":"2026-07-05T03:57:00Z"},{"alias_kind":"arxiv_version","alias_value":"2202.07414v1","created_at":"2026-07-05T03:57:00Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2202.07414","created_at":"2026-07-05T03:57:00Z"},{"alias_kind":"pith_short_12","alias_value":"AN5GUGI25DHK","created_at":"2026-07-05T03:57:00Z"},{"alias_kind":"pith_short_16","alias_value":"AN5GUGI25DHKUOIJ","created_at":"2026-07-05T03:57:00Z"},{"alias_kind":"pith_short_8","alias_value":"AN5GUGI2","created_at":"2026-07-05T03:57:00Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2022:AN5GUGI25DHKUOIJD2TZW6PST5","target":"record","payload":{"canonical_record":{"source":{"id":"2202.07414","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2022-02-15T14:04:44Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"e1904713a469f2b71e5829a20f2bacd1f7cede564b92cdeb6d671baccc205758","abstract_canon_sha256":"a17eecca836ba059c563933fdaade51d93adb31edaaf3c24ef25bbeca3b56b59"},"schema_version":"1.0"},"canonical_sha256":"037a6a191ae8ceaa39091ea79b79f29f6b3d1bcf54794347ccb895fa2c707da5","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T03:57:00.004534Z","signature_b64":"7Il7WmN5ro8vNCnEEYYMOaqxr9Ldc0raE/3jV8PboZP4F4ztUdXw9sn57CNvO2Wu0mG3ogv6nJgWHR3zzYP+Dw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"037a6a191ae8ceaa39091ea79b79f29f6b3d1bcf54794347ccb895fa2c707da5","last_reissued_at":"2026-07-05T03:57:00.004189Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T03:57:00.004189Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2202.07414","source_version":1,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T03:57:00Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"eoMwaCroyuL/EftGNIOa1+8voTlU31eTjcuvzsvCy0kNjC2PDMQlyY7InIfm0zRlTys/NknaXnoGbTsWkrr3DA==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-21T06:12:04.048260Z"},"content_sha256":"7a3550581323471c70fe1b98f87e92f3c2df8f2793cc953dcec02e966d4d61ee","schema_version":"1.0","event_id":"sha256:7a3550581323471c70fe1b98f87e92f3c2df8f2793cc953dcec02e966d4d61ee"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2022:AN5GUGI25DHKUOIJD2TZW6PST5","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Interpretable Reinforcement Learning with Multilevel Subgoal Discovery","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.AI","authors_text":"Alexander Demin, Denis Ponomaryov","submitted_at":"2022-02-15T14:04:44Z","abstract_excerpt":"We propose a novel Reinforcement Learning model for discrete environments, which is inherently interpretable and supports the discovery of deep subgoal hierarchies. In the model, an agent learns information about environment in the form of probabilistic rules, while policies for (sub)goals are learned as combinations thereof. No reward function is required for learning; an agent only needs to be given a primary goal to achieve. Subgoals of a goal G from the hierarchy are computed as descriptions of states, which if previously achieved increase the total efficiency of the available policies for"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2202.07414","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2202.07414/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T03:57:00Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"ShjA5xf9xTlgI9WtN5k+iUyC2ziqLlhBiCDW9bmtKmzWzpu4c9daHwuPXykEZQT8sHDDwNDBF3T0ZINY2YeZCA==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-21T06:12:04.048841Z"},"content_sha256":"66398497307d676ab24af51f7dc879dc1b1a2f57e7d3582dece8854a76e7b90c","schema_version":"1.0","event_id":"sha256:66398497307d676ab24af51f7dc879dc1b1a2f57e7d3582dece8854a76e7b90c"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/AN5GUGI25DHKUOIJD2TZW6PST5/bundle.json","state_url":"https://pith.science/pith/AN5GUGI25DHKUOIJD2TZW6PST5/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/AN5GUGI25DHKUOIJD2TZW6PST5/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-21T06:12:04Z","links":{"resolver":"https://pith.science/pith/AN5GUGI25DHKUOIJD2TZW6PST5","bundle":"https://pith.science/pith/AN5GUGI25DHKUOIJD2TZW6PST5/bundle.json","state":"https://pith.science/pith/AN5GUGI25DHKUOIJD2TZW6PST5/state.json","well_known_bundle":"https://pith.science/.well-known/pith/AN5GUGI25DHKUOIJD2TZW6PST5/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2022:AN5GUGI25DHKUOIJD2TZW6PST5","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"a17eecca836ba059c563933fdaade51d93adb31edaaf3c24ef25bbeca3b56b59","cross_cats_sorted":["cs.LG"],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2022-02-15T14:04:44Z","title_canon_sha256":"e1904713a469f2b71e5829a20f2bacd1f7cede564b92cdeb6d671baccc205758"},"schema_version":"1.0","source":{"id":"2202.07414","kind":"arxiv","version":1}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2202.07414","created_at":"2026-07-05T03:57:00Z"},{"alias_kind":"arxiv_version","alias_value":"2202.07414v1","created_at":"2026-07-05T03:57:00Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2202.07414","created_at":"2026-07-05T03:57:00Z"},{"alias_kind":"pith_short_12","alias_value":"AN5GUGI25DHK","created_at":"2026-07-05T03:57:00Z"},{"alias_kind":"pith_short_16","alias_value":"AN5GUGI25DHKUOIJ","created_at":"2026-07-05T03:57:00Z"},{"alias_kind":"pith_short_8","alias_value":"AN5GUGI2","created_at":"2026-07-05T03:57:00Z"}],"graph_snapshots":[{"event_id":"sha256:66398497307d676ab24af51f7dc879dc1b1a2f57e7d3582dece8854a76e7b90c","target":"graph","created_at":"2026-07-05T03:57:00Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2202.07414/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"We propose a novel Reinforcement Learning model for discrete environments, which is inherently interpretable and supports the discovery of deep subgoal hierarchies. In the model, an agent learns information about environment in the form of probabilistic rules, while policies for (sub)goals are learned as combinations thereof. No reward function is required for learning; an agent only needs to be given a primary goal to achieve. Subgoals of a goal G from the hierarchy are computed as descriptions of states, which if previously achieved increase the total efficiency of the available policies for","authors_text":"Alexander Demin, Denis Ponomaryov","cross_cats":["cs.LG"],"headline":"","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2022-02-15T14:04:44Z","title":"Interpretable Reinforcement Learning with Multilevel Subgoal Discovery"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2202.07414","kind":"arxiv","version":1},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:7a3550581323471c70fe1b98f87e92f3c2df8f2793cc953dcec02e966d4d61ee","target":"record","created_at":"2026-07-05T03:57:00Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"a17eecca836ba059c563933fdaade51d93adb31edaaf3c24ef25bbeca3b56b59","cross_cats_sorted":["cs.LG"],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2022-02-15T14:04:44Z","title_canon_sha256":"e1904713a469f2b71e5829a20f2bacd1f7cede564b92cdeb6d671baccc205758"},"schema_version":"1.0","source":{"id":"2202.07414","kind":"arxiv","version":1}},"canonical_sha256":"037a6a191ae8ceaa39091ea79b79f29f6b3d1bcf54794347ccb895fa2c707da5","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"037a6a191ae8ceaa39091ea79b79f29f6b3d1bcf54794347ccb895fa2c707da5","first_computed_at":"2026-07-05T03:57:00.004189Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T03:57:00.004189Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"7Il7WmN5ro8vNCnEEYYMOaqxr9Ldc0raE/3jV8PboZP4F4ztUdXw9sn57CNvO2Wu0mG3ogv6nJgWHR3zzYP+Dw==","signature_status":"signed_v1","signed_at":"2026-07-05T03:57:00.004534Z","signed_message":"canonical_sha256_bytes"},"source_id":"2202.07414","source_kind":"arxiv","source_version":1}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:7a3550581323471c70fe1b98f87e92f3c2df8f2793cc953dcec02e966d4d61ee","sha256:66398497307d676ab24af51f7dc879dc1b1a2f57e7d3582dece8854a76e7b90c"],"state_sha256":"ffab5424558f4795bffad1b46dce12bb6354ba49edb09076d023d713c7605a91"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"8VH+j+qYsNiHl8KSINgfe5ZkBA7tMfSAPj9D1HVMYARjS1FzKzXky4fK+1j4Nha3alFGWp/zWMDYqNRCASDwBg==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-21T06:12:04.057679Z","bundle_sha256":"b91e6e29fdfc14cecc8175963f7864d3aea5fda4a57ff1985fc22001a1ea4f8b"}}