{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2020:VV72OS5SJKZGUK2MOPLW62HZBV","short_pith_number":"pith:VV72OS5S","schema_version":"1.0","canonical_sha256":"ad7fa74bb24ab26a2b4c73d76f68f90d6c10f951d2459958a37dc5903027a1f4","source":{"kind":"arxiv","id":"2003.10113","version":1},"attestation_state":"computed","paper":{"title":"Algorithms for Non-Stationary Generalized Linear Bandits","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["stat.ML"],"primary_cat":"cs.LG","authors_text":"Aur\\'elien Garivier (UMPA-ENSL), Olivier Capp\\'e (DI-ENS), Yoan Russac (DI-ENS)","submitted_at":"2020-03-23T07:44:59Z","abstract_excerpt":"The statistical framework of Generalized Linear Models (GLM) can be applied to sequential problems involving categorical or ordinal rewards associated, for instance, with clicks, likes or ratings. In the example of binary rewards, logistic regression is well-known to be preferable to the use of standard linear modeling. Previous works have shown how to deal with GLMs in contextual online learning with bandit feedback when the environment is assumed to be stationary. In this paper, we relax this latter assumption and propose two upper confidence bound based algorithms that make use of either a "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2003.10113","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2020-03-23T07:44:59Z","cross_cats_sorted":["stat.ML"],"title_canon_sha256":"abbc666cfd19b3efaa01d75778029ad2f96443c8b6e64487f957344dba815d52","abstract_canon_sha256":"5664c4293fd07e4ff738d68534b7156c1f9c041ed8c9f0f8b644ba4d09697486"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T00:49:49.786705Z","signature_b64":"U7h8SaB9AnFB0w8CdgTlb+gzcOTsI1OZGZYG5GPpRRMNLlisrR8THk7tITzNV1cTBTbdq+2VQIL3bNo04S0WAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ad7fa74bb24ab26a2b4c73d76f68f90d6c10f951d2459958a37dc5903027a1f4","last_reissued_at":"2026-07-05T00:49:49.786336Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T00:49:49.786336Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Algorithms for Non-Stationary Generalized Linear Bandits","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["stat.ML"],"primary_cat":"cs.LG","authors_text":"Aur\\'elien Garivier (UMPA-ENSL), Olivier Capp\\'e (DI-ENS), Yoan Russac (DI-ENS)","submitted_at":"2020-03-23T07:44:59Z","abstract_excerpt":"The statistical framework of Generalized Linear Models (GLM) can be applied to sequential problems involving categorical or ordinal rewards associated, for instance, with clicks, likes or ratings. In the example of binary rewards, logistic regression is well-known to be preferable to the use of standard linear modeling. Previous works have shown how to deal with GLMs in contextual online learning with bandit feedback when the environment is assumed to be stationary. In this paper, we relax this latter assumption and propose two upper confidence bound based algorithms that make use of either a "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2003.10113","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2003.10113/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2003.10113","created_at":"2026-07-05T00:49:49.786406+00:00"},{"alias_kind":"arxiv_version","alias_value":"2003.10113v1","created_at":"2026-07-05T00:49:49.786406+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2003.10113","created_at":"2026-07-05T00:49:49.786406+00:00"},{"alias_kind":"pith_short_12","alias_value":"VV72OS5SJKZG","created_at":"2026-07-05T00:49:49.786406+00:00"},{"alias_kind":"pith_short_16","alias_value":"VV72OS5SJKZGUK2M","created_at":"2026-07-05T00:49:49.786406+00:00"},{"alias_kind":"pith_short_8","alias_value":"VV72OS5S","created_at":"2026-07-05T00:49:49.786406+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.09802","citing_title":"Bandits for Efficient Experimentation: Adapting to Control Group, Preferences, and Context Drifts","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2605.25590","citing_title":"Nonstationary Generalized Linear Bandits with Discounted Online Mirror Descent","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2605.14193","citing_title":"Equilibrium and Pricing in Consumer Networks with Nonlinear Utilities: An Online Shape-Constrained Learning Approach","ref_index":61,"is_internal_anchor":false},{"citing_arxiv_id":"2604.16684","citing_title":"DARLING: Detection Augmented Reinforcement Learning with Non-Stationary Guarantees","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2604.16684","citing_title":"DARLING: Detection Augmented Reinforcement Learning with Non-Stationary Guarantees","ref_index":8,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/VV72OS5SJKZGUK2MOPLW62HZBV","json":"https://pith.science/pith/VV72OS5SJKZGUK2MOPLW62HZBV.json","graph_json":"https://pith.science/api/pith-number/VV72OS5SJKZGUK2MOPLW62HZBV/graph.json","events_json":"https://pith.science/api/pith-number/VV72OS5SJKZGUK2MOPLW62HZBV/events.json","paper":"https://pith.science/paper/VV72OS5S"},"agent_actions":{"view_html":"https://pith.science/pith/VV72OS5SJKZGUK2MOPLW62HZBV","download_json":"https://pith.science/pith/VV72OS5SJKZGUK2MOPLW62HZBV.json","view_paper":"https://pith.science/paper/VV72OS5S","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2003.10113&json=true","fetch_graph":"https://pith.science/api/pith-number/VV72OS5SJKZGUK2MOPLW62HZBV/graph.json","fetch_events":"https://pith.science/api/pith-number/VV72OS5SJKZGUK2MOPLW62HZBV/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/VV72OS5SJKZGUK2MOPLW62HZBV/action/timestamp_anchor","attest_storage":"https://pith.science/pith/VV72OS5SJKZGUK2MOPLW62HZBV/action/storage_attestation","attest_author":"https://pith.science/pith/VV72OS5SJKZGUK2MOPLW62HZBV/action/author_attestation","sign_citation":"https://pith.science/pith/VV72OS5SJKZGUK2MOPLW62HZBV/action/citation_signature","submit_replication":"https://pith.science/pith/VV72OS5SJKZGUK2MOPLW62HZBV/action/replication_record"}},"created_at":"2026-07-05T00:49:49.786406+00:00","updated_at":"2026-07-05T00:49:49.786406+00:00"}