{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:L2B7D5NSXZQIBCINP2CCWHOZKU","short_pith_number":"pith:L2B7D5NS","schema_version":"1.0","canonical_sha256":"5e83f1f5b2be6080890d7e842b1dd9550eeee942dfd40445b810710bbfab1b6f","source":{"kind":"arxiv","id":"2101.08980","version":1},"attestation_state":"computed","paper":{"title":"Nonstationary Stochastic Multiarmed Bandits: UCB Policies and Minimax Regret","license":"http://creativecommons.org/publicdomain/zero/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Lai Wei, Vaibhav Srivastava","submitted_at":"2021-01-22T07:34:09Z","abstract_excerpt":"We study the nonstationary stochastic Multi-Armed Bandit (MAB) problem in which the distribution of rewards associated with each arm are assumed to be time-varying and the total variation in the expected rewards is subject to a variation budget. The regret of a policy is defined by the difference in the expected cumulative rewards obtained using the policy and using an oracle that selects the arm with the maximum mean reward at each time. We characterize the performance of the proposed policies in terms of the worst-case regret, which is the supremum of the regret over the set of reward distri"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2101.08980","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/publicdomain/zero/1.0/","primary_cat":"cs.LG","submitted_at":"2021-01-22T07:34:09Z","cross_cats_sorted":[],"title_canon_sha256":"3224cbce17f960dfc111f05b15b293324fba2139d605e2c7e18d5a1b726f0210","abstract_canon_sha256":"8fb84aa3e727c8ee2239f0d0cb510d0df9aad5638a659f83327429888d9738d5"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T02:08:52.628369Z","signature_b64":"QQuT36+aCL8auYWw7gFFubJNmbMPvmD8xTylaODCsHuwTqekhFrKQ4qajVex/PsAdoFWuPFrUyfeAq3QWdoSBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"5e83f1f5b2be6080890d7e842b1dd9550eeee942dfd40445b810710bbfab1b6f","last_reissued_at":"2026-07-05T02:08:52.627884Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T02:08:52.627884Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Nonstationary Stochastic Multiarmed Bandits: UCB Policies and Minimax Regret","license":"http://creativecommons.org/publicdomain/zero/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Lai Wei, Vaibhav Srivastava","submitted_at":"2021-01-22T07:34:09Z","abstract_excerpt":"We study the nonstationary stochastic Multi-Armed Bandit (MAB) problem in which the distribution of rewards associated with each arm are assumed to be time-varying and the total variation in the expected rewards is subject to a variation budget. The regret of a policy is defined by the difference in the expected cumulative rewards obtained using the policy and using an oracle that selects the arm with the maximum mean reward at each time. We characterize the performance of the proposed policies in terms of the worst-case regret, which is the supremum of the regret over the set of reward distri"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2101.08980","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2101.08980/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2101.08980","created_at":"2026-07-05T02:08:52.627950+00:00"},{"alias_kind":"arxiv_version","alias_value":"2101.08980v1","created_at":"2026-07-05T02:08:52.627950+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2101.08980","created_at":"2026-07-05T02:08:52.627950+00:00"},{"alias_kind":"pith_short_12","alias_value":"L2B7D5NSXZQI","created_at":"2026-07-05T02:08:52.627950+00:00"},{"alias_kind":"pith_short_16","alias_value":"L2B7D5NSXZQIBCIN","created_at":"2026-07-05T02:08:52.627950+00:00"},{"alias_kind":"pith_short_8","alias_value":"L2B7D5NS","created_at":"2026-07-05T02:08:52.627950+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2502.01000","citing_title":"Adapting Foundation Models for Few-Shot Medical Image Segmentation: Actively and Sequentially","ref_index":22,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/L2B7D5NSXZQIBCINP2CCWHOZKU","json":"https://pith.science/pith/L2B7D5NSXZQIBCINP2CCWHOZKU.json","graph_json":"https://pith.science/api/pith-number/L2B7D5NSXZQIBCINP2CCWHOZKU/graph.json","events_json":"https://pith.science/api/pith-number/L2B7D5NSXZQIBCINP2CCWHOZKU/events.json","paper":"https://pith.science/paper/L2B7D5NS"},"agent_actions":{"view_html":"https://pith.science/pith/L2B7D5NSXZQIBCINP2CCWHOZKU","download_json":"https://pith.science/pith/L2B7D5NSXZQIBCINP2CCWHOZKU.json","view_paper":"https://pith.science/paper/L2B7D5NS","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2101.08980&json=true","fetch_graph":"https://pith.science/api/pith-number/L2B7D5NSXZQIBCINP2CCWHOZKU/graph.json","fetch_events":"https://pith.science/api/pith-number/L2B7D5NSXZQIBCINP2CCWHOZKU/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/L2B7D5NSXZQIBCINP2CCWHOZKU/action/timestamp_anchor","attest_storage":"https://pith.science/pith/L2B7D5NSXZQIBCINP2CCWHOZKU/action/storage_attestation","attest_author":"https://pith.science/pith/L2B7D5NSXZQIBCINP2CCWHOZKU/action/author_attestation","sign_citation":"https://pith.science/pith/L2B7D5NSXZQIBCINP2CCWHOZKU/action/citation_signature","submit_replication":"https://pith.science/pith/L2B7D5NSXZQIBCINP2CCWHOZKU/action/replication_record"}},"created_at":"2026-07-05T02:08:52.627950+00:00","updated_at":"2026-07-05T02:08:52.627950+00:00"}