{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2019:7L5MO3JSZCJQTMEB67V55YUUQN","short_pith_number":"pith:7L5MO3JS","schema_version":"1.0","canonical_sha256":"fafac76d32c89309b081f7ebdee2948367bbe0cd027290c8b1792ca742b5692f","source":{"kind":"arxiv","id":"1910.05654","version":1},"attestation_state":"computed","paper":{"title":"Thompson Sampling in Non-Episodic Restless Bandits","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["stat.ML"],"primary_cat":"cs.LG","authors_text":"Ambuj Tewari, Marc Abeille, Young Hun Jung","submitted_at":"2019-10-12T22:30:24Z","abstract_excerpt":"Restless bandit problems assume time-varying reward distributions of the arms, which adds flexibility to the model but makes the analysis more challenging. We study learning algorithms over the unknown reward distributions and prove a sub-linear, $O(\\sqrt{T}\\log T)$, regret bound for a variant of Thompson sampling. Our analysis applies in the infinite time horizon setting, resolving the open question raised by Jung and Tewari (2019) whose analysis is limited to the episodic case. We adopt their policy mapping framework, which allows our algorithm to be efficient and simultaneously keeps the re"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"1910.05654","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2019-10-12T22:30:24Z","cross_cats_sorted":["stat.ML"],"title_canon_sha256":"3f86ac8407d1160af1060ed776f55d39285a062909ea56cac3818cf9ce7c1abf","abstract_canon_sha256":"93e589e00e0363af1e22f36ac42106258e47ab0c0c1d9f086479e39e7d3c6ad1"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T00:11:40.681172Z","signature_b64":"miARhm+C4NAjRQrGxa1oZg/zLhIYRWjIzGrx4RZAH/mQqM1ydE9olHR8D6s+ik/FslMk8MJjS4r41BxmmjJPBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"fafac76d32c89309b081f7ebdee2948367bbe0cd027290c8b1792ca742b5692f","last_reissued_at":"2026-07-05T00:11:40.680706Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T00:11:40.680706Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Thompson Sampling in Non-Episodic Restless Bandits","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["stat.ML"],"primary_cat":"cs.LG","authors_text":"Ambuj Tewari, Marc Abeille, Young Hun Jung","submitted_at":"2019-10-12T22:30:24Z","abstract_excerpt":"Restless bandit problems assume time-varying reward distributions of the arms, which adds flexibility to the model but makes the analysis more challenging. We study learning algorithms over the unknown reward distributions and prove a sub-linear, $O(\\sqrt{T}\\log T)$, regret bound for a variant of Thompson sampling. Our analysis applies in the infinite time horizon setting, resolving the open question raised by Jung and Tewari (2019) whose analysis is limited to the episodic case. We adopt their policy mapping framework, which allows our algorithm to be efficient and simultaneously keeps the re"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1910.05654","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/1910.05654/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"1910.05654","created_at":"2026-07-05T00:11:40.680764+00:00"},{"alias_kind":"arxiv_version","alias_value":"1910.05654v1","created_at":"2026-07-05T00:11:40.680764+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1910.05654","created_at":"2026-07-05T00:11:40.680764+00:00"},{"alias_kind":"pith_short_12","alias_value":"7L5MO3JSZCJQ","created_at":"2026-07-05T00:11:40.680764+00:00"},{"alias_kind":"pith_short_16","alias_value":"7L5MO3JSZCJQTMEB","created_at":"2026-07-05T00:11:40.680764+00:00"},{"alias_kind":"pith_short_8","alias_value":"7L5MO3JS","created_at":"2026-07-05T00:11:40.680764+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.27448","citing_title":"Learning in Markovian bandits with non-observable states and constrained decision epochs","ref_index":3,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/7L5MO3JSZCJQTMEB67V55YUUQN","json":"https://pith.science/pith/7L5MO3JSZCJQTMEB67V55YUUQN.json","graph_json":"https://pith.science/api/pith-number/7L5MO3JSZCJQTMEB67V55YUUQN/graph.json","events_json":"https://pith.science/api/pith-number/7L5MO3JSZCJQTMEB67V55YUUQN/events.json","paper":"https://pith.science/paper/7L5MO3JS"},"agent_actions":{"view_html":"https://pith.science/pith/7L5MO3JSZCJQTMEB67V55YUUQN","download_json":"https://pith.science/pith/7L5MO3JSZCJQTMEB67V55YUUQN.json","view_paper":"https://pith.science/paper/7L5MO3JS","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=1910.05654&json=true","fetch_graph":"https://pith.science/api/pith-number/7L5MO3JSZCJQTMEB67V55YUUQN/graph.json","fetch_events":"https://pith.science/api/pith-number/7L5MO3JSZCJQTMEB67V55YUUQN/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/7L5MO3JSZCJQTMEB67V55YUUQN/action/timestamp_anchor","attest_storage":"https://pith.science/pith/7L5MO3JSZCJQTMEB67V55YUUQN/action/storage_attestation","attest_author":"https://pith.science/pith/7L5MO3JSZCJQTMEB67V55YUUQN/action/author_attestation","sign_citation":"https://pith.science/pith/7L5MO3JSZCJQTMEB67V55YUUQN/action/citation_signature","submit_replication":"https://pith.science/pith/7L5MO3JSZCJQTMEB67V55YUUQN/action/replication_record"}},"created_at":"2026-07-05T00:11:40.680764+00:00","updated_at":"2026-07-05T00:11:40.680764+00:00"}