{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:ELTBHBPZ3EHPX5CWMUHOVDWE3F","short_pith_number":"pith:ELTBHBPZ","schema_version":"1.0","canonical_sha256":"22e61385f9d90efbf456650eea8ec4d95d3e31f5f1de03fa6cbe63450e0cee99","source":{"kind":"arxiv","id":"2112.03097","version":1},"attestation_state":"computed","paper":{"title":"Flexible Option Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Doina Precup, Martin Klissarov","submitted_at":"2021-12-06T15:07:48Z","abstract_excerpt":"Temporal abstraction in reinforcement learning (RL), offers the promise of improving generalization and knowledge transfer in complex environments, by propagating information more efficiently over time. Although option learning was initially formulated in a way that allows updating many options simultaneously, using off-policy, intra-option learning (Sutton, Precup & Singh, 1999), many of the recent hierarchical reinforcement learning approaches only update a single option at a time: the option currently executing. We revisit and extend intra-option learning in the context of deep reinforcemen"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2112.03097","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2021-12-06T15:07:48Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"7d19365a30fc4fdf2ba3c210758d843c673cbcb4b75f047dc781ebe24e1c0fd3","abstract_canon_sha256":"a83bba8629ef6bc12b3fc31e464c5c7fab52ee73af8951f23985e5e8f3f9a54c"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T03:38:01.915980Z","signature_b64":"Idf6ocsMXCby51eqEUL1m+sRMCkqB8PCJH4drJFK+HbCuVq+ELu+fITfEbwiRVbuMyZoqXL7jUeSaZ7wqYX0Ag==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"22e61385f9d90efbf456650eea8ec4d95d3e31f5f1de03fa6cbe63450e0cee99","last_reissued_at":"2026-07-05T03:38:01.915441Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T03:38:01.915441Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Flexible Option Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Doina Precup, Martin Klissarov","submitted_at":"2021-12-06T15:07:48Z","abstract_excerpt":"Temporal abstraction in reinforcement learning (RL), offers the promise of improving generalization and knowledge transfer in complex environments, by propagating information more efficiently over time. Although option learning was initially formulated in a way that allows updating many options simultaneously, using off-policy, intra-option learning (Sutton, Precup & Singh, 1999), many of the recent hierarchical reinforcement learning approaches only update a single option at a time: the option currently executing. We revisit and extend intra-option learning in the context of deep reinforcemen"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2112.03097","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2112.03097/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2112.03097","created_at":"2026-07-05T03:38:01.915511+00:00"},{"alias_kind":"arxiv_version","alias_value":"2112.03097v1","created_at":"2026-07-05T03:38:01.915511+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2112.03097","created_at":"2026-07-05T03:38:01.915511+00:00"},{"alias_kind":"pith_short_12","alias_value":"ELTBHBPZ3EHP","created_at":"2026-07-05T03:38:01.915511+00:00"},{"alias_kind":"pith_short_16","alias_value":"ELTBHBPZ3EHPX5CW","created_at":"2026-07-05T03:38:01.915511+00:00"},{"alias_kind":"pith_short_8","alias_value":"ELTBHBPZ","created_at":"2026-07-05T03:38:01.915511+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2604.20156","citing_title":"Temporally Extended Mixture-of-Experts Models","ref_index":22,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ELTBHBPZ3EHPX5CWMUHOVDWE3F","json":"https://pith.science/pith/ELTBHBPZ3EHPX5CWMUHOVDWE3F.json","graph_json":"https://pith.science/api/pith-number/ELTBHBPZ3EHPX5CWMUHOVDWE3F/graph.json","events_json":"https://pith.science/api/pith-number/ELTBHBPZ3EHPX5CWMUHOVDWE3F/events.json","paper":"https://pith.science/paper/ELTBHBPZ"},"agent_actions":{"view_html":"https://pith.science/pith/ELTBHBPZ3EHPX5CWMUHOVDWE3F","download_json":"https://pith.science/pith/ELTBHBPZ3EHPX5CWMUHOVDWE3F.json","view_paper":"https://pith.science/paper/ELTBHBPZ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2112.03097&json=true","fetch_graph":"https://pith.science/api/pith-number/ELTBHBPZ3EHPX5CWMUHOVDWE3F/graph.json","fetch_events":"https://pith.science/api/pith-number/ELTBHBPZ3EHPX5CWMUHOVDWE3F/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ELTBHBPZ3EHPX5CWMUHOVDWE3F/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ELTBHBPZ3EHPX5CWMUHOVDWE3F/action/storage_attestation","attest_author":"https://pith.science/pith/ELTBHBPZ3EHPX5CWMUHOVDWE3F/action/author_attestation","sign_citation":"https://pith.science/pith/ELTBHBPZ3EHPX5CWMUHOVDWE3F/action/citation_signature","submit_replication":"https://pith.science/pith/ELTBHBPZ3EHPX5CWMUHOVDWE3F/action/replication_record"}},"created_at":"2026-07-05T03:38:01.915511+00:00","updated_at":"2026-07-05T03:38:01.915511+00:00"}