{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:UUCP6PEI5VZXWYJZMP5GTP7LKP","short_pith_number":"pith:UUCP6PEI","schema_version":"1.0","canonical_sha256":"a504ff3c88ed737b613963fa69bfeb53d5a092619a29f56637ee5ebf306909a0","source":{"kind":"arxiv","id":"2201.04735","version":2},"attestation_state":"computed","paper":{"title":"Planning in Observable POMDPs in Quasipolynomial Time","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.DS","math.OC","stat.ML"],"primary_cat":"cs.LG","authors_text":"Ankur Moitra, Dhruv Rohatgi, Noah Golowich","submitted_at":"2022-01-12T23:16:37Z","abstract_excerpt":"Partially Observable Markov Decision Processes (POMDPs) are a natural and general model in reinforcement learning that take into account the agent's uncertainty about its current state. In the literature on POMDPs, it is customary to assume access to a planning oracle that computes an optimal policy when the parameters are known, even though the problem is known to be computationally hard. Almost all existing planning algorithms either run in exponential time, lack provable performance guarantees, or require placing strong assumptions on the transition dynamics under every possible policy. In "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2201.04735","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2022-01-12T23:16:37Z","cross_cats_sorted":["cs.DS","math.OC","stat.ML"],"title_canon_sha256":"8d81224a244934877fba1e32ca170e567de7926fb06cb0f40562eafb60c05028","abstract_canon_sha256":"46a101eade2cd8a0d3b24d98a9fc2db70ada886ec6b7988956ad03c3402bd905"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T04:07:43.479857Z","signature_b64":"/0wr3Tqm7QPfn0ub4VhkRr4cyNGHLRyAMF8rHPbgeUFnVhimHfrRuGJ2nfVH6+GPcZlZzsak2jZaP/tNQg90Ag==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"a504ff3c88ed737b613963fa69bfeb53d5a092619a29f56637ee5ebf306909a0","last_reissued_at":"2026-07-05T04:07:43.479338Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T04:07:43.479338Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Planning in Observable POMDPs in Quasipolynomial Time","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.DS","math.OC","stat.ML"],"primary_cat":"cs.LG","authors_text":"Ankur Moitra, Dhruv Rohatgi, Noah Golowich","submitted_at":"2022-01-12T23:16:37Z","abstract_excerpt":"Partially Observable Markov Decision Processes (POMDPs) are a natural and general model in reinforcement learning that take into account the agent's uncertainty about its current state. In the literature on POMDPs, it is customary to assume access to a planning oracle that computes an optimal policy when the parameters are known, even though the problem is known to be computationally hard. Almost all existing planning algorithms either run in exponential time, lack provable performance guarantees, or require placing strong assumptions on the transition dynamics under every possible policy. In "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2201.04735","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2201.04735/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2201.04735","created_at":"2026-07-05T04:07:43.479403+00:00"},{"alias_kind":"arxiv_version","alias_value":"2201.04735v2","created_at":"2026-07-05T04:07:43.479403+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2201.04735","created_at":"2026-07-05T04:07:43.479403+00:00"},{"alias_kind":"pith_short_12","alias_value":"UUCP6PEI5VZX","created_at":"2026-07-05T04:07:43.479403+00:00"},{"alias_kind":"pith_short_16","alias_value":"UUCP6PEI5VZXWYJZ","created_at":"2026-07-05T04:07:43.479403+00:00"},{"alias_kind":"pith_short_8","alias_value":"UUCP6PEI","created_at":"2026-07-05T04:07:43.479403+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.02363","citing_title":"Minimax-Optimal Policy Regret in Partially Observable Markov Games","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2605.00160","citing_title":"Approximations and Learning for Decentralized Stochastic Control and Near Optimal Finite Window Policies","ref_index":17,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/UUCP6PEI5VZXWYJZMP5GTP7LKP","json":"https://pith.science/pith/UUCP6PEI5VZXWYJZMP5GTP7LKP.json","graph_json":"https://pith.science/api/pith-number/UUCP6PEI5VZXWYJZMP5GTP7LKP/graph.json","events_json":"https://pith.science/api/pith-number/UUCP6PEI5VZXWYJZMP5GTP7LKP/events.json","paper":"https://pith.science/paper/UUCP6PEI"},"agent_actions":{"view_html":"https://pith.science/pith/UUCP6PEI5VZXWYJZMP5GTP7LKP","download_json":"https://pith.science/pith/UUCP6PEI5VZXWYJZMP5GTP7LKP.json","view_paper":"https://pith.science/paper/UUCP6PEI","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2201.04735&json=true","fetch_graph":"https://pith.science/api/pith-number/UUCP6PEI5VZXWYJZMP5GTP7LKP/graph.json","fetch_events":"https://pith.science/api/pith-number/UUCP6PEI5VZXWYJZMP5GTP7LKP/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/UUCP6PEI5VZXWYJZMP5GTP7LKP/action/timestamp_anchor","attest_storage":"https://pith.science/pith/UUCP6PEI5VZXWYJZMP5GTP7LKP/action/storage_attestation","attest_author":"https://pith.science/pith/UUCP6PEI5VZXWYJZMP5GTP7LKP/action/author_attestation","sign_citation":"https://pith.science/pith/UUCP6PEI5VZXWYJZMP5GTP7LKP/action/citation_signature","submit_replication":"https://pith.science/pith/UUCP6PEI5VZXWYJZMP5GTP7LKP/action/replication_record"}},"created_at":"2026-07-05T04:07:43.479403+00:00","updated_at":"2026-07-05T04:07:43.479403+00:00"}