{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:242KN2W23UZ4TN5HTPW6UKAV2A","short_pith_number":"pith:242KN2W2","schema_version":"1.0","canonical_sha256":"d734a6eadadd33c9b7a79bedea2815d01f19988237cf2cfcd7d43202c3820122","source":{"kind":"arxiv","id":"2209.14997","version":3},"attestation_state":"computed","paper":{"title":"Optimistic MLE -- A Generic Model-based Algorithm for Partially Observable Sequential Decision Making","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","stat.ML"],"primary_cat":"cs.LG","authors_text":"Chi Jin, Csaba Szepesv\\'ari, Praneeth Netrapalli, Qinghua Liu","submitted_at":"2022-09-29T17:56:25Z","abstract_excerpt":"This paper introduces a simple efficient learning algorithms for general sequential decision making. The algorithm combines Optimism for exploration with Maximum Likelihood Estimation for model estimation, which is thus named OMLE. We prove that OMLE learns the near-optimal policies of an enormously rich class of sequential decision making problems in a polynomial number of samples. This rich class includes not only a majority of known tractable model-based Reinforcement Learning (RL) problems (such as tabular MDPs, factored MDPs, low witness rank problems, tabular weakly-revealing/observable "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2209.14997","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2022-09-29T17:56:25Z","cross_cats_sorted":["cs.AI","stat.ML"],"title_canon_sha256":"1115c38a35c24b4472800180f8198a27fcd1c2d43f411c30262c9503cb45906f","abstract_canon_sha256":"11b1e2a1c250f95a08ed6edffaa5c984e88dd9693228d4a04c82e4379a761791"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:18:49.933064Z","signature_b64":"u0Z7sG6wKUIK9W7TrEiNyDcubFvD5V+HrQJStI7e2279nVnmZMMn14+t6qmnLzD6MLJYQxkB2qHFjjUTYxIEAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"d734a6eadadd33c9b7a79bedea2815d01f19988237cf2cfcd7d43202c3820122","last_reissued_at":"2026-07-05T05:18:49.932635Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:18:49.932635Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Optimistic MLE -- A Generic Model-based Algorithm for Partially Observable Sequential Decision Making","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","stat.ML"],"primary_cat":"cs.LG","authors_text":"Chi Jin, Csaba Szepesv\\'ari, Praneeth Netrapalli, Qinghua Liu","submitted_at":"2022-09-29T17:56:25Z","abstract_excerpt":"This paper introduces a simple efficient learning algorithms for general sequential decision making. The algorithm combines Optimism for exploration with Maximum Likelihood Estimation for model estimation, which is thus named OMLE. We prove that OMLE learns the near-optimal policies of an enormously rich class of sequential decision making problems in a polynomial number of samples. This rich class includes not only a majority of known tractable model-based Reinforcement Learning (RL) problems (such as tabular MDPs, factored MDPs, low witness rank problems, tabular weakly-revealing/observable "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2209.14997","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2209.14997/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2209.14997","created_at":"2026-07-05T05:18:49.932691+00:00"},{"alias_kind":"arxiv_version","alias_value":"2209.14997v3","created_at":"2026-07-05T05:18:49.932691+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2209.14997","created_at":"2026-07-05T05:18:49.932691+00:00"},{"alias_kind":"pith_short_12","alias_value":"242KN2W23UZ4","created_at":"2026-07-05T05:18:49.932691+00:00"},{"alias_kind":"pith_short_16","alias_value":"242KN2W23UZ4TN5H","created_at":"2026-07-05T05:18:49.932691+00:00"},{"alias_kind":"pith_short_8","alias_value":"242KN2W2","created_at":"2026-07-05T05:18:49.932691+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.01242","citing_title":"Breaking the Computational Barrier: Provably Efficient Actor-Critic for Low-Rank MDPs","ref_index":78,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/242KN2W23UZ4TN5HTPW6UKAV2A","json":"https://pith.science/pith/242KN2W23UZ4TN5HTPW6UKAV2A.json","graph_json":"https://pith.science/api/pith-number/242KN2W23UZ4TN5HTPW6UKAV2A/graph.json","events_json":"https://pith.science/api/pith-number/242KN2W23UZ4TN5HTPW6UKAV2A/events.json","paper":"https://pith.science/paper/242KN2W2"},"agent_actions":{"view_html":"https://pith.science/pith/242KN2W23UZ4TN5HTPW6UKAV2A","download_json":"https://pith.science/pith/242KN2W23UZ4TN5HTPW6UKAV2A.json","view_paper":"https://pith.science/paper/242KN2W2","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2209.14997&json=true","fetch_graph":"https://pith.science/api/pith-number/242KN2W23UZ4TN5HTPW6UKAV2A/graph.json","fetch_events":"https://pith.science/api/pith-number/242KN2W23UZ4TN5HTPW6UKAV2A/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/242KN2W23UZ4TN5HTPW6UKAV2A/action/timestamp_anchor","attest_storage":"https://pith.science/pith/242KN2W23UZ4TN5HTPW6UKAV2A/action/storage_attestation","attest_author":"https://pith.science/pith/242KN2W23UZ4TN5HTPW6UKAV2A/action/author_attestation","sign_citation":"https://pith.science/pith/242KN2W23UZ4TN5HTPW6UKAV2A/action/citation_signature","submit_replication":"https://pith.science/pith/242KN2W23UZ4TN5HTPW6UKAV2A/action/replication_record"}},"created_at":"2026-07-05T05:18:49.932691+00:00","updated_at":"2026-07-05T05:18:49.932691+00:00"}