{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:6SAJZHIH5TIBYEUXGIFYJ6PCGN","short_pith_number":"pith:6SAJZHIH","schema_version":"1.0","canonical_sha256":"f4809c9d07ecd01c1297320b84f9e2334e90c4fa0eff55e4f411705144d94396","source":{"kind":"arxiv","id":"2302.07186","version":2},"attestation_state":"computed","paper":{"title":"Adversarial Rewards in Universal Learning for Contextual Bandits","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG","math.ST","stat.TH"],"primary_cat":"stat.ML","authors_text":"Moise Blanchard, Patrick Jaillet, Steve Hanneke","submitted_at":"2023-02-14T16:54:22Z","abstract_excerpt":"We study the fundamental limits of learning in contextual bandits, where a learner's rewards depend on their actions and a known context, which extends the canonical multi-armed bandit to the case where side-information is available. We are interested in universally consistent algorithms, which achieve sublinear regret compared to any measurable fixed policy, without any function class restriction. For stationary contextual bandits, when the underlying reward mechanism is time-invariant, Blanchard et. al (2022) characterized learnable context processes for which universal consistency is achiev"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2302.07186","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"stat.ML","submitted_at":"2023-02-14T16:54:22Z","cross_cats_sorted":["cs.LG","math.ST","stat.TH"],"title_canon_sha256":"88a9ed1228b29feedba8582885069746e7f66fb48618e285253337c1c34c3f62","abstract_canon_sha256":"8faee0b467d16db55c049b073602f9b47e7bcab9e5fe108e97048f639dc5adfc"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:19:39.221336Z","signature_b64":"F5goa9q28UYZyO+xLesD4qEJgoK+sHlPzHQhL8Obt6rIkijyyIoIIajlc61//W9xT7VH6NshZSuHILt4+pQ/CQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f4809c9d07ecd01c1297320b84f9e2334e90c4fa0eff55e4f411705144d94396","last_reissued_at":"2026-07-05T06:19:39.220870Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:19:39.220870Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Adversarial Rewards in Universal Learning for Contextual Bandits","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG","math.ST","stat.TH"],"primary_cat":"stat.ML","authors_text":"Moise Blanchard, Patrick Jaillet, Steve Hanneke","submitted_at":"2023-02-14T16:54:22Z","abstract_excerpt":"We study the fundamental limits of learning in contextual bandits, where a learner's rewards depend on their actions and a known context, which extends the canonical multi-armed bandit to the case where side-information is available. We are interested in universally consistent algorithms, which achieve sublinear regret compared to any measurable fixed policy, without any function class restriction. For stationary contextual bandits, when the underlying reward mechanism is time-invariant, Blanchard et. al (2022) characterized learnable context processes for which universal consistency is achiev"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2302.07186","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2302.07186/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2302.07186","created_at":"2026-07-05T06:19:39.220928+00:00"},{"alias_kind":"arxiv_version","alias_value":"2302.07186v2","created_at":"2026-07-05T06:19:39.220928+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2302.07186","created_at":"2026-07-05T06:19:39.220928+00:00"},{"alias_kind":"pith_short_12","alias_value":"6SAJZHIH5TIB","created_at":"2026-07-05T06:19:39.220928+00:00"},{"alias_kind":"pith_short_16","alias_value":"6SAJZHIH5TIBYEUX","created_at":"2026-07-05T06:19:39.220928+00:00"},{"alias_kind":"pith_short_8","alias_value":"6SAJZHIH","created_at":"2026-07-05T06:19:39.220928+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2502.08759","citing_title":"Contextual bandits with entropy-based human feedback","ref_index":2022,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/6SAJZHIH5TIBYEUXGIFYJ6PCGN","json":"https://pith.science/pith/6SAJZHIH5TIBYEUXGIFYJ6PCGN.json","graph_json":"https://pith.science/api/pith-number/6SAJZHIH5TIBYEUXGIFYJ6PCGN/graph.json","events_json":"https://pith.science/api/pith-number/6SAJZHIH5TIBYEUXGIFYJ6PCGN/events.json","paper":"https://pith.science/paper/6SAJZHIH"},"agent_actions":{"view_html":"https://pith.science/pith/6SAJZHIH5TIBYEUXGIFYJ6PCGN","download_json":"https://pith.science/pith/6SAJZHIH5TIBYEUXGIFYJ6PCGN.json","view_paper":"https://pith.science/paper/6SAJZHIH","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2302.07186&json=true","fetch_graph":"https://pith.science/api/pith-number/6SAJZHIH5TIBYEUXGIFYJ6PCGN/graph.json","fetch_events":"https://pith.science/api/pith-number/6SAJZHIH5TIBYEUXGIFYJ6PCGN/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/6SAJZHIH5TIBYEUXGIFYJ6PCGN/action/timestamp_anchor","attest_storage":"https://pith.science/pith/6SAJZHIH5TIBYEUXGIFYJ6PCGN/action/storage_attestation","attest_author":"https://pith.science/pith/6SAJZHIH5TIBYEUXGIFYJ6PCGN/action/author_attestation","sign_citation":"https://pith.science/pith/6SAJZHIH5TIBYEUXGIFYJ6PCGN/action/citation_signature","submit_replication":"https://pith.science/pith/6SAJZHIH5TIBYEUXGIFYJ6PCGN/action/replication_record"}},"created_at":"2026-07-05T06:19:39.220928+00:00","updated_at":"2026-07-05T06:19:39.220928+00:00"}