{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:1999:DFSDBPEZWAPNQ6ZCN5EOKQFKIW","short_pith_number":"pith:DFSDBPEZ","canonical_record":{"source":{"id":"cs/9905014","kind":"arxiv","version":1},"metadata":{"license":"","primary_cat":"cs.LG","submitted_at":"1999-05-21T14:26:07Z","cross_cats_sorted":[],"title_canon_sha256":"79b9e16d1a24abed13e985ef0a20a3bb27be7526bcf74d9432a9a06f89aceb74","abstract_canon_sha256":"d522467e1d05423f6338fbc9a34ace43e0059253825cc8ef6314f55f375ecde0"},"schema_version":"1.0"},"canonical_sha256":"196430bc99b01ed87b226f48e540aa459ce41beff881cc5b94902509b60ebf27","source":{"kind":"arxiv","id":"cs/9905014","version":1},"source_aliases":[{"alias_kind":"arxiv","alias_value":"cs/9905014","created_at":"2026-07-04T14:25:10Z"},{"alias_kind":"arxiv_version","alias_value":"cs/9905014v1","created_at":"2026-07-04T14:25:10Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.cs/9905014","created_at":"2026-07-04T14:25:10Z"},{"alias_kind":"pith_short_12","alias_value":"DFSDBPEZWAPN","created_at":"2026-07-04T14:25:10Z"},{"alias_kind":"pith_short_16","alias_value":"DFSDBPEZWAPNQ6ZC","created_at":"2026-07-04T14:25:10Z"},{"alias_kind":"pith_short_8","alias_value":"DFSDBPEZ","created_at":"2026-07-04T14:25:10Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:1999:DFSDBPEZWAPNQ6ZCN5EOKQFKIW","target":"record","payload":{"canonical_record":{"source":{"id":"cs/9905014","kind":"arxiv","version":1},"metadata":{"license":"","primary_cat":"cs.LG","submitted_at":"1999-05-21T14:26:07Z","cross_cats_sorted":[],"title_canon_sha256":"79b9e16d1a24abed13e985ef0a20a3bb27be7526bcf74d9432a9a06f89aceb74","abstract_canon_sha256":"d522467e1d05423f6338fbc9a34ace43e0059253825cc8ef6314f55f375ecde0"},"schema_version":"1.0"},"canonical_sha256":"196430bc99b01ed87b226f48e540aa459ce41beff881cc5b94902509b60ebf27","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-04T14:25:10.795620Z","signature_b64":"a2Qbx49OR6QsoGzAVllmohmi35p6rrtTp55Tk96/lqhP/jwmlCzrXVXdava1PQ9IirOmi9C9+KKCnmwfsjQEAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"196430bc99b01ed87b226f48e540aa459ce41beff881cc5b94902509b60ebf27","last_reissued_at":"2026-07-04T14:25:10.795230Z","signature_status":"signed_v1","first_computed_at":"2026-07-04T14:25:10.795230Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"cs/9905014","source_version":1,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-04T14:25:10Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"kzEXHBjUXuSif8vbUdTuWy34dos39IDOzqc6CY/euuzCqe1EewLBq/4DrrtCwDFqKW5qEhpwGi68VBg0jbtoDw==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-10T23:30:58.793842Z"},"content_sha256":"d0a94e4d284e6edb1a438b78eee3239efe940c187a9822dd33e22d8a18ec22e2","schema_version":"1.0","event_id":"sha256:d0a94e4d284e6edb1a438b78eee3239efe940c187a9822dd33e22d8a18ec22e2"},{"event_type":"graph_snapshot","subject_pith_number":"pith:1999:DFSDBPEZWAPNQ6ZCN5EOKQFKIW","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Hierarchical Reinforcement Learning with the MAXQ Value Function Decomposition","license":"","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Thomas G. Dietterich","submitted_at":"1999-05-21T14:26:07Z","abstract_excerpt":"This paper presents the MAXQ approach to hierarchical reinforcement learning based on decomposing the target Markov decision process (MDP) into a hierarchy of smaller MDPs and decomposing the value function of the target MDP into an additive combination of the value functions of the smaller MDPs. The paper defines the MAXQ hierarchy, proves formal results on its representational power, and establishes five conditions for the safe use of state abstractions. The paper presents an online model-free learning algorithm, MAXQ-Q, and proves that it converges wih probability 1 to a kind of locally-opt"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"cs/9905014","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/cs/9905014/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-04T14:25:10Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"+fEucDkJdZaDbna4QvXRsZi0AB7pTi0VG3nMlJ7N+ASLrFgXF2c+QwnV6o9tKFvxx1mI8Ggor9Xyky6U+/zxDQ==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-10T23:30:58.794625Z"},"content_sha256":"98235c9ca940bd109397875849101d155df1fa190546ae1d6943d10ee1b7c67e","schema_version":"1.0","event_id":"sha256:98235c9ca940bd109397875849101d155df1fa190546ae1d6943d10ee1b7c67e"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/DFSDBPEZWAPNQ6ZCN5EOKQFKIW/bundle.json","state_url":"https://pith.science/pith/DFSDBPEZWAPNQ6ZCN5EOKQFKIW/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/DFSDBPEZWAPNQ6ZCN5EOKQFKIW/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-10T23:30:58Z","links":{"resolver":"https://pith.science/pith/DFSDBPEZWAPNQ6ZCN5EOKQFKIW","bundle":"https://pith.science/pith/DFSDBPEZWAPNQ6ZCN5EOKQFKIW/bundle.json","state":"https://pith.science/pith/DFSDBPEZWAPNQ6ZCN5EOKQFKIW/state.json","well_known_bundle":"https://pith.science/.well-known/pith/DFSDBPEZWAPNQ6ZCN5EOKQFKIW/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:1999:DFSDBPEZWAPNQ6ZCN5EOKQFKIW","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"d522467e1d05423f6338fbc9a34ace43e0059253825cc8ef6314f55f375ecde0","cross_cats_sorted":[],"license":"","primary_cat":"cs.LG","submitted_at":"1999-05-21T14:26:07Z","title_canon_sha256":"79b9e16d1a24abed13e985ef0a20a3bb27be7526bcf74d9432a9a06f89aceb74"},"schema_version":"1.0","source":{"id":"cs/9905014","kind":"arxiv","version":1}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"cs/9905014","created_at":"2026-07-04T14:25:10Z"},{"alias_kind":"arxiv_version","alias_value":"cs/9905014v1","created_at":"2026-07-04T14:25:10Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.cs/9905014","created_at":"2026-07-04T14:25:10Z"},{"alias_kind":"pith_short_12","alias_value":"DFSDBPEZWAPN","created_at":"2026-07-04T14:25:10Z"},{"alias_kind":"pith_short_16","alias_value":"DFSDBPEZWAPNQ6ZC","created_at":"2026-07-04T14:25:10Z"},{"alias_kind":"pith_short_8","alias_value":"DFSDBPEZ","created_at":"2026-07-04T14:25:10Z"}],"graph_snapshots":[{"event_id":"sha256:98235c9ca940bd109397875849101d155df1fa190546ae1d6943d10ee1b7c67e","target":"graph","created_at":"2026-07-04T14:25:10Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/cs/9905014/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"This paper presents the MAXQ approach to hierarchical reinforcement learning based on decomposing the target Markov decision process (MDP) into a hierarchy of smaller MDPs and decomposing the value function of the target MDP into an additive combination of the value functions of the smaller MDPs. The paper defines the MAXQ hierarchy, proves formal results on its representational power, and establishes five conditions for the safe use of state abstractions. The paper presents an online model-free learning algorithm, MAXQ-Q, and proves that it converges wih probability 1 to a kind of locally-opt","authors_text":"Thomas G. Dietterich","cross_cats":[],"headline":"","license":"","primary_cat":"cs.LG","submitted_at":"1999-05-21T14:26:07Z","title":"Hierarchical Reinforcement Learning with the MAXQ Value Function Decomposition"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"cs/9905014","kind":"arxiv","version":1},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:d0a94e4d284e6edb1a438b78eee3239efe940c187a9822dd33e22d8a18ec22e2","target":"record","created_at":"2026-07-04T14:25:10Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"d522467e1d05423f6338fbc9a34ace43e0059253825cc8ef6314f55f375ecde0","cross_cats_sorted":[],"license":"","primary_cat":"cs.LG","submitted_at":"1999-05-21T14:26:07Z","title_canon_sha256":"79b9e16d1a24abed13e985ef0a20a3bb27be7526bcf74d9432a9a06f89aceb74"},"schema_version":"1.0","source":{"id":"cs/9905014","kind":"arxiv","version":1}},"canonical_sha256":"196430bc99b01ed87b226f48e540aa459ce41beff881cc5b94902509b60ebf27","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"196430bc99b01ed87b226f48e540aa459ce41beff881cc5b94902509b60ebf27","first_computed_at":"2026-07-04T14:25:10.795230Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-04T14:25:10.795230Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"a2Qbx49OR6QsoGzAVllmohmi35p6rrtTp55Tk96/lqhP/jwmlCzrXVXdava1PQ9IirOmi9C9+KKCnmwfsjQEAw==","signature_status":"signed_v1","signed_at":"2026-07-04T14:25:10.795620Z","signed_message":"canonical_sha256_bytes"},"source_id":"cs/9905014","source_kind":"arxiv","source_version":1}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:d0a94e4d284e6edb1a438b78eee3239efe940c187a9822dd33e22d8a18ec22e2","sha256:98235c9ca940bd109397875849101d155df1fa190546ae1d6943d10ee1b7c67e"],"state_sha256":"fbe9899edfe94704d9f9f697a93eacfa97c98f64390e2e67700463a23d89799b"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"6Y8vawNc16TjQORIRpemwXhx3eFKolV3Zz0sdGgIJ4uI3pC/EUYVXb4pcg2iz+J7JTsdokEBSe4myI6TeqpmBA==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-10T23:30:58.799196Z","bundle_sha256":"ab9b4e81ff720e5ff3da203116c92358cc5ae7db0472a841d389e55fec78aafc"}}