{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2025:3X4PHGRETULAKAW7G5RJRGK4TC","short_pith_number":"pith:3X4PHGRE","canonical_record":{"source":{"id":"2507.04373","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-07-06T12:38:42Z","cross_cats_sorted":[],"title_canon_sha256":"c03ce38a3eb6e5b4e39e8e9a015448fc06af833b49008795a9c7b555a5d7af84","abstract_canon_sha256":"1307d1033ca6f3cd5cc93be3d5b0baaa4b1151f14b71ba3ad47cccf9ea616bbf"},"schema_version":"1.0"},"canonical_sha256":"ddf8f39a249d160502df376298995c98944f5fd6f5e8cd1f427998b6fea45587","source":{"kind":"arxiv","id":"2507.04373","version":1},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2507.04373","created_at":"2026-07-05T11:32:46Z"},{"alias_kind":"arxiv_version","alias_value":"2507.04373v1","created_at":"2026-07-05T11:32:46Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2507.04373","created_at":"2026-07-05T11:32:46Z"},{"alias_kind":"pith_short_12","alias_value":"3X4PHGRETULA","created_at":"2026-07-05T11:32:46Z"},{"alias_kind":"pith_short_16","alias_value":"3X4PHGRETULAKAW7","created_at":"2026-07-05T11:32:46Z"},{"alias_kind":"pith_short_8","alias_value":"3X4PHGRE","created_at":"2026-07-05T11:32:46Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2025:3X4PHGRETULAKAW7G5RJRGK4TC","target":"record","payload":{"canonical_record":{"source":{"id":"2507.04373","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-07-06T12:38:42Z","cross_cats_sorted":[],"title_canon_sha256":"c03ce38a3eb6e5b4e39e8e9a015448fc06af833b49008795a9c7b555a5d7af84","abstract_canon_sha256":"1307d1033ca6f3cd5cc93be3d5b0baaa4b1151f14b71ba3ad47cccf9ea616bbf"},"schema_version":"1.0"},"canonical_sha256":"ddf8f39a249d160502df376298995c98944f5fd6f5e8cd1f427998b6fea45587","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:32:46.756477Z","signature_b64":"hpNkOTpE6L8IUdQLltbJApjbK/neVzFm8WXbde7MtqLpBPxSbr9KWdjWJb/yX8sUy2KxVrxSsQVxPX2sj0KICg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ddf8f39a249d160502df376298995c98944f5fd6f5e8cd1f427998b6fea45587","last_reissued_at":"2026-07-05T11:32:46.755963Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:32:46.755963Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2507.04373","source_version":1,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T11:32:46Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"2FytLDRrG9uusNUgdvX5Qg65uoOQdIY2aX2LWxEVBUY1QK0icM61tmK7+N4BlM73nkQgXAp/xyVc5eHQFQ01Dw==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-07T03:42:01.108576Z"},"content_sha256":"62ff292e2985648b6b47186a039c2c85041d55b8660d52f61480cf866b195f6f","schema_version":"1.0","event_id":"sha256:62ff292e2985648b6b47186a039c2c85041d55b8660d52f61480cf866b195f6f"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2025:3X4PHGRETULAKAW7G5RJRGK4TC","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Hierarchical Reinforcement Learning with Targeted Causal Interventions","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Matthias Grossglauser, Negar Kiyavash, Saber Salehkaleybar, Sadegh Khorasani","submitted_at":"2025-07-06T12:38:42Z","abstract_excerpt":"Hierarchical reinforcement learning (HRL) improves the efficiency of long-horizon reinforcement-learning tasks with sparse rewards by decomposing the task into a hierarchy of subgoals. The main challenge of HRL is efficient discovery of the hierarchical structure among subgoals and utilizing this structure to achieve the final goal. We address this challenge by modeling the subgoal structure as a causal graph and propose a causal discovery algorithm to learn it. Additionally, rather than intervening on the subgoals at random during exploration, we harness the discovered causal model to priorit"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2507.04373","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2507.04373/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T11:32:46Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"7CArhNs3wPucnbhzDxN5BKc9LvM3ZKykgfTgMNELP1nRQCtVMMlZQ2NsfymxRkGKMD4PwHwjXDjTP5Y4VtTzAw==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-07T03:42:01.109108Z"},"content_sha256":"61e3b0dffcfecf514017e8bacceac7f7a33ce357c8b40197ae313d23df833e4d","schema_version":"1.0","event_id":"sha256:61e3b0dffcfecf514017e8bacceac7f7a33ce357c8b40197ae313d23df833e4d"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/3X4PHGRETULAKAW7G5RJRGK4TC/bundle.json","state_url":"https://pith.science/pith/3X4PHGRETULAKAW7G5RJRGK4TC/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/3X4PHGRETULAKAW7G5RJRGK4TC/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-07T03:42:01Z","links":{"resolver":"https://pith.science/pith/3X4PHGRETULAKAW7G5RJRGK4TC","bundle":"https://pith.science/pith/3X4PHGRETULAKAW7G5RJRGK4TC/bundle.json","state":"https://pith.science/pith/3X4PHGRETULAKAW7G5RJRGK4TC/state.json","well_known_bundle":"https://pith.science/.well-known/pith/3X4PHGRETULAKAW7G5RJRGK4TC/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2025:3X4PHGRETULAKAW7G5RJRGK4TC","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"1307d1033ca6f3cd5cc93be3d5b0baaa4b1151f14b71ba3ad47cccf9ea616bbf","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-07-06T12:38:42Z","title_canon_sha256":"c03ce38a3eb6e5b4e39e8e9a015448fc06af833b49008795a9c7b555a5d7af84"},"schema_version":"1.0","source":{"id":"2507.04373","kind":"arxiv","version":1}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2507.04373","created_at":"2026-07-05T11:32:46Z"},{"alias_kind":"arxiv_version","alias_value":"2507.04373v1","created_at":"2026-07-05T11:32:46Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2507.04373","created_at":"2026-07-05T11:32:46Z"},{"alias_kind":"pith_short_12","alias_value":"3X4PHGRETULA","created_at":"2026-07-05T11:32:46Z"},{"alias_kind":"pith_short_16","alias_value":"3X4PHGRETULAKAW7","created_at":"2026-07-05T11:32:46Z"},{"alias_kind":"pith_short_8","alias_value":"3X4PHGRE","created_at":"2026-07-05T11:32:46Z"}],"graph_snapshots":[{"event_id":"sha256:61e3b0dffcfecf514017e8bacceac7f7a33ce357c8b40197ae313d23df833e4d","target":"graph","created_at":"2026-07-05T11:32:46Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2507.04373/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Hierarchical reinforcement learning (HRL) improves the efficiency of long-horizon reinforcement-learning tasks with sparse rewards by decomposing the task into a hierarchy of subgoals. The main challenge of HRL is efficient discovery of the hierarchical structure among subgoals and utilizing this structure to achieve the final goal. We address this challenge by modeling the subgoal structure as a causal graph and propose a causal discovery algorithm to learn it. Additionally, rather than intervening on the subgoals at random during exploration, we harness the discovered causal model to priorit","authors_text":"Matthias Grossglauser, Negar Kiyavash, Saber Salehkaleybar, Sadegh Khorasani","cross_cats":[],"headline":"","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-07-06T12:38:42Z","title":"Hierarchical Reinforcement Learning with Targeted Causal Interventions"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2507.04373","kind":"arxiv","version":1},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:62ff292e2985648b6b47186a039c2c85041d55b8660d52f61480cf866b195f6f","target":"record","created_at":"2026-07-05T11:32:46Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"1307d1033ca6f3cd5cc93be3d5b0baaa4b1151f14b71ba3ad47cccf9ea616bbf","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-07-06T12:38:42Z","title_canon_sha256":"c03ce38a3eb6e5b4e39e8e9a015448fc06af833b49008795a9c7b555a5d7af84"},"schema_version":"1.0","source":{"id":"2507.04373","kind":"arxiv","version":1}},"canonical_sha256":"ddf8f39a249d160502df376298995c98944f5fd6f5e8cd1f427998b6fea45587","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"ddf8f39a249d160502df376298995c98944f5fd6f5e8cd1f427998b6fea45587","first_computed_at":"2026-07-05T11:32:46.755963Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T11:32:46.755963Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"hpNkOTpE6L8IUdQLltbJApjbK/neVzFm8WXbde7MtqLpBPxSbr9KWdjWJb/yX8sUy2KxVrxSsQVxPX2sj0KICg==","signature_status":"signed_v1","signed_at":"2026-07-05T11:32:46.756477Z","signed_message":"canonical_sha256_bytes"},"source_id":"2507.04373","source_kind":"arxiv","source_version":1}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:62ff292e2985648b6b47186a039c2c85041d55b8660d52f61480cf866b195f6f","sha256:61e3b0dffcfecf514017e8bacceac7f7a33ce357c8b40197ae313d23df833e4d"],"state_sha256":"2ca18a637c7367e102abc0fe6437631a511356fcea54ab8c8cbc63afad353083"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"zzFmZUNs300of0clQuHsoGqzpA/oSlEj+91lOU8vjVGpg6gHwYa8DLZQMg+qGGKoSwByFjQVC0TSRy8zD94SDQ==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-07T03:42:01.112916Z","bundle_sha256":"8dabc687b8a788b09e89825027a334239ef9013b2ca4bca525db3905761fd870"}}