{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:SUDKTVIPTJ7D7W5PPEOLEZOXPN","short_pith_number":"pith:SUDKTVIP","schema_version":"1.0","canonical_sha256":"9506a9d50f9a7e3fdbaf791cb265d77b76e0c846f851980dc9d3d5b17c4931f5","source":{"kind":"arxiv","id":"2502.08259","version":2},"attestation_state":"computed","paper":{"title":"Balancing optimism and pessimism in offline-to-online learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Csaba Szepesvari, Flore Sentenac, Ilbin Lee","submitted_at":"2025-02-12T10:05:25Z","abstract_excerpt":"We consider what we call the offline-to-online learning setting, focusing on stochastic finite-armed bandit problems. In offline-to-online learning, a learner starts with offline data collected from interactions with an unknown environment in a way that is not under the learner's control. Given this data, the learner begins interacting with the environment, gradually improving its initial strategy as it collects more data to maximize its total reward. The learner in this setting faces a fundamental dilemma: if the policy is deployed for only a short period, a suitable strategy (in a number of "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.08259","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-02-12T10:05:25Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"c389cdb376f6d6835b6eaab6bd44859244c6e890e1663329137dc42a5f030dec","abstract_canon_sha256":"3278fe94cf83bf6d157a9ff501ed31ff8a1efbd44f9a6661622ff48744db0b3c"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:27:25.033328Z","signature_b64":"UlTHYQlIJ78sLAOjPG8iomUYb0nRj2g/w35L1rlmFgP6QHT44hiElTtKDFDWC0ZASjteO9IHTXV22a6CxIGFAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"9506a9d50f9a7e3fdbaf791cb265d77b76e0c846f851980dc9d3d5b17c4931f5","last_reissued_at":"2026-07-05T10:27:25.032313Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:27:25.032313Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Balancing optimism and pessimism in offline-to-online learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Csaba Szepesvari, Flore Sentenac, Ilbin Lee","submitted_at":"2025-02-12T10:05:25Z","abstract_excerpt":"We consider what we call the offline-to-online learning setting, focusing on stochastic finite-armed bandit problems. In offline-to-online learning, a learner starts with offline data collected from interactions with an unknown environment in a way that is not under the learner's control. Given this data, the learner begins interacting with the environment, gradually improving its initial strategy as it collects more data to maximize its total reward. The learner in this setting faces a fundamental dilemma: if the policy is deployed for only a short period, a suitable strategy (in a number of "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.08259","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.08259/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.08259","created_at":"2026-07-05T10:27:25.032485+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.08259v2","created_at":"2026-07-05T10:27:25.032485+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.08259","created_at":"2026-07-05T10:27:25.032485+00:00"},{"alias_kind":"pith_short_12","alias_value":"SUDKTVIPTJ7D","created_at":"2026-07-05T10:27:25.032485+00:00"},{"alias_kind":"pith_short_16","alias_value":"SUDKTVIPTJ7D7W5P","created_at":"2026-07-05T10:27:25.032485+00:00"},{"alias_kind":"pith_short_8","alias_value":"SUDKTVIP","created_at":"2026-07-05T10:27:25.032485+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.22579","citing_title":"Stationary Robust Mean-Field Games under Model Mismatches","ref_index":61,"is_internal_anchor":false},{"citing_arxiv_id":"2606.04305","citing_title":"Offline-to-Online Learning in Linear Bandits","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2605.25789","citing_title":"On the Benefits of Free Exploration for Regret Minimization in Multi-Armed Bandits","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2512.04341","citing_title":"Long-Horizon Model-Based Offline Reinforcement Learning Without Explicit Conservatism","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10289","citing_title":"Sample-Mean Anchored Thompson Sampling for Offline-to-Online Learning with Distribution Shift","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10289","citing_title":"Sample-Mean Anchored Thompson Sampling for Offline-to-Online Learning with Distribution Shift","ref_index":35,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/SUDKTVIPTJ7D7W5PPEOLEZOXPN","json":"https://pith.science/pith/SUDKTVIPTJ7D7W5PPEOLEZOXPN.json","graph_json":"https://pith.science/api/pith-number/SUDKTVIPTJ7D7W5PPEOLEZOXPN/graph.json","events_json":"https://pith.science/api/pith-number/SUDKTVIPTJ7D7W5PPEOLEZOXPN/events.json","paper":"https://pith.science/paper/SUDKTVIP"},"agent_actions":{"view_html":"https://pith.science/pith/SUDKTVIPTJ7D7W5PPEOLEZOXPN","download_json":"https://pith.science/pith/SUDKTVIPTJ7D7W5PPEOLEZOXPN.json","view_paper":"https://pith.science/paper/SUDKTVIP","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.08259&json=true","fetch_graph":"https://pith.science/api/pith-number/SUDKTVIPTJ7D7W5PPEOLEZOXPN/graph.json","fetch_events":"https://pith.science/api/pith-number/SUDKTVIPTJ7D7W5PPEOLEZOXPN/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/SUDKTVIPTJ7D7W5PPEOLEZOXPN/action/timestamp_anchor","attest_storage":"https://pith.science/pith/SUDKTVIPTJ7D7W5PPEOLEZOXPN/action/storage_attestation","attest_author":"https://pith.science/pith/SUDKTVIPTJ7D7W5PPEOLEZOXPN/action/author_attestation","sign_citation":"https://pith.science/pith/SUDKTVIPTJ7D7W5PPEOLEZOXPN/action/citation_signature","submit_replication":"https://pith.science/pith/SUDKTVIPTJ7D7W5PPEOLEZOXPN/action/replication_record"}},"created_at":"2026-07-05T10:27:25.032485+00:00","updated_at":"2026-07-05T10:27:25.032485+00:00"}