{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:42MA7PXW2FCZXO43BCPGG6KQ3M","short_pith_number":"pith:42MA7PXW","schema_version":"1.0","canonical_sha256":"e6980fbef6d1459bbb9b089e637950db2c6122b6da0b743353d57171599167a3","source":{"kind":"arxiv","id":"2101.11992","version":4},"attestation_state":"computed","paper":{"title":"Acting in Delayed Environments with Non-Stationary Markov Policies","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Esther Derman, Gal Dalal, Shie Mannor","submitted_at":"2021-01-28T13:35:37Z","abstract_excerpt":"The standard Markov Decision Process (MDP) formulation hinges on the assumption that an action is executed immediately after it was chosen. However, assuming it is often unrealistic and can lead to catastrophic failures in applications such as robotic manipulation, cloud computing, and finance. We introduce a framework for learning and planning in MDPs where the decision-maker commits actions that are executed with a delay of $m$ steps. The brute-force state augmentation baseline where the state is concatenated to the last $m$ committed actions suffers from an exponential complexity in $m$, as"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2101.11992","kind":"arxiv","version":4},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2021-01-28T13:35:37Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"babf7c560990b33c8b3306766af160e47d177481fdbd17edd75f8aeebb2354fb","abstract_canon_sha256":"456510909619327fd8ad7cec63ebb46754bb295ef80d7a8b8d691cdbc654eda1"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:23:21.289892Z","signature_b64":"+s/DELrGanZQ4bwzITiV/PzPMPHlLlOUYOFSCsIwSJB3wq4G6JBBACQa4uu6PpXm9hEK1DBA3K0bluJhdk9XCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"e6980fbef6d1459bbb9b089e637950db2c6122b6da0b743353d57171599167a3","last_reissued_at":"2026-07-05T07:23:21.289438Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:23:21.289438Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Acting in Delayed Environments with Non-Stationary Markov Policies","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Esther Derman, Gal Dalal, Shie Mannor","submitted_at":"2021-01-28T13:35:37Z","abstract_excerpt":"The standard Markov Decision Process (MDP) formulation hinges on the assumption that an action is executed immediately after it was chosen. However, assuming it is often unrealistic and can lead to catastrophic failures in applications such as robotic manipulation, cloud computing, and finance. We introduce a framework for learning and planning in MDPs where the decision-maker commits actions that are executed with a delay of $m$ steps. The brute-force state augmentation baseline where the state is concatenated to the last $m$ committed actions suffers from an exponential complexity in $m$, as"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2101.11992","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2101.11992/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2101.11992","created_at":"2026-07-05T07:23:21.289498+00:00"},{"alias_kind":"arxiv_version","alias_value":"2101.11992v4","created_at":"2026-07-05T07:23:21.289498+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2101.11992","created_at":"2026-07-05T07:23:21.289498+00:00"},{"alias_kind":"pith_short_12","alias_value":"42MA7PXW2FCZ","created_at":"2026-07-05T07:23:21.289498+00:00"},{"alias_kind":"pith_short_16","alias_value":"42MA7PXW2FCZXO43","created_at":"2026-07-05T07:23:21.289498+00:00"},{"alias_kind":"pith_short_8","alias_value":"42MA7PXW","created_at":"2026-07-05T07:23:21.289498+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.26463","citing_title":"Finding the Time to Think: Learning Planning Budgets in Real-Time RL","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2606.26463","citing_title":"Finding the Time to Think: Learning Planning Budgets in Real-Time RL","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2509.20869","citing_title":"Model-Based Reinforcement Learning under Random Observation Delays","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2604.03641","citing_title":"Delayed homomorphic reinforcement learning for environments with delayed feedback","ref_index":1,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/42MA7PXW2FCZXO43BCPGG6KQ3M","json":"https://pith.science/pith/42MA7PXW2FCZXO43BCPGG6KQ3M.json","graph_json":"https://pith.science/api/pith-number/42MA7PXW2FCZXO43BCPGG6KQ3M/graph.json","events_json":"https://pith.science/api/pith-number/42MA7PXW2FCZXO43BCPGG6KQ3M/events.json","paper":"https://pith.science/paper/42MA7PXW"},"agent_actions":{"view_html":"https://pith.science/pith/42MA7PXW2FCZXO43BCPGG6KQ3M","download_json":"https://pith.science/pith/42MA7PXW2FCZXO43BCPGG6KQ3M.json","view_paper":"https://pith.science/paper/42MA7PXW","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2101.11992&json=true","fetch_graph":"https://pith.science/api/pith-number/42MA7PXW2FCZXO43BCPGG6KQ3M/graph.json","fetch_events":"https://pith.science/api/pith-number/42MA7PXW2FCZXO43BCPGG6KQ3M/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/42MA7PXW2FCZXO43BCPGG6KQ3M/action/timestamp_anchor","attest_storage":"https://pith.science/pith/42MA7PXW2FCZXO43BCPGG6KQ3M/action/storage_attestation","attest_author":"https://pith.science/pith/42MA7PXW2FCZXO43BCPGG6KQ3M/action/author_attestation","sign_citation":"https://pith.science/pith/42MA7PXW2FCZXO43BCPGG6KQ3M/action/citation_signature","submit_replication":"https://pith.science/pith/42MA7PXW2FCZXO43BCPGG6KQ3M/action/replication_record"}},"created_at":"2026-07-05T07:23:21.289498+00:00","updated_at":"2026-07-05T07:23:21.289498+00:00"}