{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2017:BUM2NKX7JRETSRVCGIISCUXS7Q","short_pith_number":"pith:BUM2NKX7","schema_version":"1.0","canonical_sha256":"0d19a6aaff4c493946a232112152f2fc038890a468ccfafb65d80bbe7ce68228","source":{"kind":"arxiv","id":"1712.00948","version":5},"attestation_state":"computed","paper":{"title":"Learning Multi-Level Hierarchies with Hindsight","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG","cs.NE","cs.RO"],"primary_cat":"cs.AI","authors_text":"Andrew Levy, George Konidaris, Kate Saenko, Robert Platt","submitted_at":"2017-12-04T08:18:08Z","abstract_excerpt":"Hierarchical agents have the potential to solve sequential decision making tasks with greater sample efficiency than their non-hierarchical counterparts because hierarchical agents can break down tasks into sets of subtasks that only require short sequences of decisions. In order to realize this potential of faster learning, hierarchical agents need to be able to learn their multiple levels of policies in parallel so these simpler subproblems can be solved simultaneously. Yet, learning multiple levels of policies in parallel is hard because it is inherently unstable: changes in a policy at one"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"1712.00948","kind":"arxiv","version":5},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2017-12-04T08:18:08Z","cross_cats_sorted":["cs.LG","cs.NE","cs.RO"],"title_canon_sha256":"fbb10d01a5dd5d055135f026fe7ffd3db0a63cd9ac5daae36b6ba68e8802d215","abstract_canon_sha256":"272203b977bd49d27e458055bb93729c3db164afea82f10525e6e95e11ff6eca"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T00:02:10.773568Z","signature_b64":"lpq/uiqNtARs1FR6uHGkRLM/BnbZnh7ChlNazlUOe5eRKnw6fmCIWefVa2KDAsccwMXvM1d7teYNIMXlCJQIBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"0d19a6aaff4c493946a232112152f2fc038890a468ccfafb65d80bbe7ce68228","last_reissued_at":"2026-07-05T00:02:10.773082Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T00:02:10.773082Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Learning Multi-Level Hierarchies with Hindsight","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG","cs.NE","cs.RO"],"primary_cat":"cs.AI","authors_text":"Andrew Levy, George Konidaris, Kate Saenko, Robert Platt","submitted_at":"2017-12-04T08:18:08Z","abstract_excerpt":"Hierarchical agents have the potential to solve sequential decision making tasks with greater sample efficiency than their non-hierarchical counterparts because hierarchical agents can break down tasks into sets of subtasks that only require short sequences of decisions. In order to realize this potential of faster learning, hierarchical agents need to be able to learn their multiple levels of policies in parallel so these simpler subproblems can be solved simultaneously. Yet, learning multiple levels of policies in parallel is hard because it is inherently unstable: changes in a policy at one"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1712.00948","kind":"arxiv","version":5},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/1712.00948/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"1712.00948","created_at":"2026-07-05T00:02:10.773138+00:00"},{"alias_kind":"arxiv_version","alias_value":"1712.00948v5","created_at":"2026-07-05T00:02:10.773138+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1712.00948","created_at":"2026-07-05T00:02:10.773138+00:00"},{"alias_kind":"pith_short_12","alias_value":"BUM2NKX7JRET","created_at":"2026-07-05T00:02:10.773138+00:00"},{"alias_kind":"pith_short_16","alias_value":"BUM2NKX7JRETSRVC","created_at":"2026-07-05T00:02:10.773138+00:00"},{"alias_kind":"pith_short_8","alias_value":"BUM2NKX7","created_at":"2026-07-05T00:02:10.773138+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":8,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.09476","citing_title":"Goal Sets, Not Goal States: Queryable Robot Goals through Goal-Set Hindsight Relabeling","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"1907.00664","citing_title":"Learning World Graphs to Accelerate Hierarchical Reinforcement Learning","ref_index":57,"is_internal_anchor":false},{"citing_arxiv_id":"2605.22711","citing_title":"Abstraction for Offline Goal-Conditioned Reinforcement Learning","ref_index":37,"is_internal_anchor":false},{"citing_arxiv_id":"2506.21039","citing_title":"Strict Subgoal Execution: Reliable Long-Horizon Planning in Hierarchical Reinforcement Learning","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2509.00338","citing_title":"Scalable Option Learning in High-Throughput Environments","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2409.12917","citing_title":"Training Language Models to Self-Correct via Reinforcement Learning","ref_index":217,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12261","citing_title":"Delay-Empowered Causal Hierarchical Reinforcement Learning","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01862","citing_title":"QHyer: Q-conditioned Hybrid Attention-mamba Transformer for Offline Goal-conditioned RL","ref_index":101,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/BUM2NKX7JRETSRVCGIISCUXS7Q","json":"https://pith.science/pith/BUM2NKX7JRETSRVCGIISCUXS7Q.json","graph_json":"https://pith.science/api/pith-number/BUM2NKX7JRETSRVCGIISCUXS7Q/graph.json","events_json":"https://pith.science/api/pith-number/BUM2NKX7JRETSRVCGIISCUXS7Q/events.json","paper":"https://pith.science/paper/BUM2NKX7"},"agent_actions":{"view_html":"https://pith.science/pith/BUM2NKX7JRETSRVCGIISCUXS7Q","download_json":"https://pith.science/pith/BUM2NKX7JRETSRVCGIISCUXS7Q.json","view_paper":"https://pith.science/paper/BUM2NKX7","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=1712.00948&json=true","fetch_graph":"https://pith.science/api/pith-number/BUM2NKX7JRETSRVCGIISCUXS7Q/graph.json","fetch_events":"https://pith.science/api/pith-number/BUM2NKX7JRETSRVCGIISCUXS7Q/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/BUM2NKX7JRETSRVCGIISCUXS7Q/action/timestamp_anchor","attest_storage":"https://pith.science/pith/BUM2NKX7JRETSRVCGIISCUXS7Q/action/storage_attestation","attest_author":"https://pith.science/pith/BUM2NKX7JRETSRVCGIISCUXS7Q/action/author_attestation","sign_citation":"https://pith.science/pith/BUM2NKX7JRETSRVCGIISCUXS7Q/action/citation_signature","submit_replication":"https://pith.science/pith/BUM2NKX7JRETSRVCGIISCUXS7Q/action/replication_record"}},"created_at":"2026-07-05T00:02:10.773138+00:00","updated_at":"2026-07-05T00:02:10.773138+00:00"}