{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:MQ65W4JCSYEQ7K7L7S4NIASHDC","short_pith_number":"pith:MQ65W4JC","schema_version":"1.0","canonical_sha256":"643ddb712296090fabebfcb8d4024718a3f2896dcc8d026c88e5bd0d29ad58ca","source":{"kind":"arxiv","id":"2310.10107","version":4},"attestation_state":"computed","paper":{"title":"Posterior Sampling-based Online Learning for Episodic POMDPs","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":["cs.AI","cs.SY","eess.SY","stat.ML"],"primary_cat":"cs.LG","authors_text":"Ashutosh Nayyar, Dengwang Tang, Dongze Ye, Pierluigi Nuzzo, Rahul Jain","submitted_at":"2023-10-16T06:41:13Z","abstract_excerpt":"Learning in POMDPs is known to be significantly harder than in MDPs. In this paper, we consider the online learning problem for episodic POMDPs with unknown transition and observation models. We propose a Posterior Sampling-based reinforcement learning algorithm for POMDPs (PS4POMDPs), which is much simpler and more implementable compared to state-of-the-art optimism-based online learning algorithms for POMDPs. We show that the Bayesian regret of the proposed algorithm scales as the square root of the number of episodes and is polynomial in the other parameters. In a general setting, the regre"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2310.10107","kind":"arxiv","version":4},"metadata":{"license":"http://creativecommons.org/licenses/by-sa/4.0/","primary_cat":"cs.LG","submitted_at":"2023-10-16T06:41:13Z","cross_cats_sorted":["cs.AI","cs.SY","eess.SY","stat.ML"],"title_canon_sha256":"50c404f41f031aa045a68c3875cc49f7047ac8d23746038b915dded11f47eee2","abstract_canon_sha256":"6359f5e38eb43dc4550b70b267dcbfd87672e265da996f80e79f7f25d16ea0c7"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:24:31.589682Z","signature_b64":"b5v80h8nB5P/JXh3UmqvNGLPyfGqaWQNfYiAPvDEXyS5Vee4vZBX+Gv1kWvDm3uw2OWM9TPP3fmFJ53Qz7mlDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"643ddb712296090fabebfcb8d4024718a3f2896dcc8d026c88e5bd0d29ad58ca","last_reissued_at":"2026-07-05T09:24:31.589116Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:24:31.589116Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Posterior Sampling-based Online Learning for Episodic POMDPs","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":["cs.AI","cs.SY","eess.SY","stat.ML"],"primary_cat":"cs.LG","authors_text":"Ashutosh Nayyar, Dengwang Tang, Dongze Ye, Pierluigi Nuzzo, Rahul Jain","submitted_at":"2023-10-16T06:41:13Z","abstract_excerpt":"Learning in POMDPs is known to be significantly harder than in MDPs. In this paper, we consider the online learning problem for episodic POMDPs with unknown transition and observation models. We propose a Posterior Sampling-based reinforcement learning algorithm for POMDPs (PS4POMDPs), which is much simpler and more implementable compared to state-of-the-art optimism-based online learning algorithms for POMDPs. We show that the Bayesian regret of the proposed algorithm scales as the square root of the number of episodes and is polynomial in the other parameters. In a general setting, the regre"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2310.10107","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2310.10107/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2310.10107","created_at":"2026-07-05T09:24:31.589181+00:00"},{"alias_kind":"arxiv_version","alias_value":"2310.10107v4","created_at":"2026-07-05T09:24:31.589181+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2310.10107","created_at":"2026-07-05T09:24:31.589181+00:00"},{"alias_kind":"pith_short_12","alias_value":"MQ65W4JCSYEQ","created_at":"2026-07-05T09:24:31.589181+00:00"},{"alias_kind":"pith_short_16","alias_value":"MQ65W4JCSYEQ7K7L","created_at":"2026-07-05T09:24:31.589181+00:00"},{"alias_kind":"pith_short_8","alias_value":"MQ65W4JC","created_at":"2026-07-05T09:24:31.589181+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/MQ65W4JCSYEQ7K7L7S4NIASHDC","json":"https://pith.science/pith/MQ65W4JCSYEQ7K7L7S4NIASHDC.json","graph_json":"https://pith.science/api/pith-number/MQ65W4JCSYEQ7K7L7S4NIASHDC/graph.json","events_json":"https://pith.science/api/pith-number/MQ65W4JCSYEQ7K7L7S4NIASHDC/events.json","paper":"https://pith.science/paper/MQ65W4JC"},"agent_actions":{"view_html":"https://pith.science/pith/MQ65W4JCSYEQ7K7L7S4NIASHDC","download_json":"https://pith.science/pith/MQ65W4JCSYEQ7K7L7S4NIASHDC.json","view_paper":"https://pith.science/paper/MQ65W4JC","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2310.10107&json=true","fetch_graph":"https://pith.science/api/pith-number/MQ65W4JCSYEQ7K7L7S4NIASHDC/graph.json","fetch_events":"https://pith.science/api/pith-number/MQ65W4JCSYEQ7K7L7S4NIASHDC/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/MQ65W4JCSYEQ7K7L7S4NIASHDC/action/timestamp_anchor","attest_storage":"https://pith.science/pith/MQ65W4JCSYEQ7K7L7S4NIASHDC/action/storage_attestation","attest_author":"https://pith.science/pith/MQ65W4JCSYEQ7K7L7S4NIASHDC/action/author_attestation","sign_citation":"https://pith.science/pith/MQ65W4JCSYEQ7K7L7S4NIASHDC/action/citation_signature","submit_replication":"https://pith.science/pith/MQ65W4JCSYEQ7K7L7S4NIASHDC/action/replication_record"}},"created_at":"2026-07-05T09:24:31.589181+00:00","updated_at":"2026-07-05T09:24:31.589181+00:00"}