{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2018:P6KWOMMXTWA7GLJSV6HT2TJM73","short_pith_number":"pith:P6KWOMMX","schema_version":"1.0","canonical_sha256":"7f956731979d81f32d32af8f3d4d2cfefb893411b2faa67692457276b22e8427","source":{"kind":"arxiv","id":"1802.04064","version":5},"attestation_state":"computed","paper":{"title":"A Contextual Bandit Bake-off","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"stat.ML","authors_text":"Alberto Bietti, Alekh Agarwal, John Langford","submitted_at":"2018-02-12T14:30:56Z","abstract_excerpt":"Contextual bandit algorithms are essential for solving many real-world interactive machine learning problems. Despite multiple recent successes on statistically and computationally efficient methods, the practical behavior of these algorithms is still poorly understood. We leverage the availability of large numbers of supervised learning datasets to empirically evaluate contextual bandit algorithms, focusing on practical methods that learn by relying on optimization oracles from supervised learning. We find that a recent method (Foster et al., 2018) using optimism under uncertainty works the b"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"1802.04064","kind":"arxiv","version":5},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"stat.ML","submitted_at":"2018-02-12T14:30:56Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"00b8e8bd4e8e6d6c24bc6ccb0a3beaf3a19084119d386a80e8f7fe82cb07ff67","abstract_canon_sha256":"9da63e26942c448df80fa5635b5c70a0afbde73756b8e0d5aa144db59dc5e9c0"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T02:46:22.211280Z","signature_b64":"GGO/dZeAEpyK/84EthdRkah89e6OAiaYkRzZ12zA4/lc9+cELa+LHmdUYN9Y2Feg5QlJS6U/kT/3JwHlzMEjCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"7f956731979d81f32d32af8f3d4d2cfefb893411b2faa67692457276b22e8427","last_reissued_at":"2026-07-05T02:46:22.210803Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T02:46:22.210803Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"A Contextual Bandit Bake-off","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"stat.ML","authors_text":"Alberto Bietti, Alekh Agarwal, John Langford","submitted_at":"2018-02-12T14:30:56Z","abstract_excerpt":"Contextual bandit algorithms are essential for solving many real-world interactive machine learning problems. Despite multiple recent successes on statistically and computationally efficient methods, the practical behavior of these algorithms is still poorly understood. We leverage the availability of large numbers of supervised learning datasets to empirically evaluate contextual bandit algorithms, focusing on practical methods that learn by relying on optimization oracles from supervised learning. We find that a recent method (Foster et al., 2018) using optimism under uncertainty works the b"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1802.04064","kind":"arxiv","version":5},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/1802.04064/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"1802.04064","created_at":"2026-07-05T02:46:22.210872+00:00"},{"alias_kind":"arxiv_version","alias_value":"1802.04064v5","created_at":"2026-07-05T02:46:22.210872+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1802.04064","created_at":"2026-07-05T02:46:22.210872+00:00"},{"alias_kind":"pith_short_12","alias_value":"P6KWOMMXTWA7","created_at":"2026-07-05T02:46:22.210872+00:00"},{"alias_kind":"pith_short_16","alias_value":"P6KWOMMXTWA7GLJS","created_at":"2026-07-05T02:46:22.210872+00:00"},{"alias_kind":"pith_short_8","alias_value":"P6KWOMMX","created_at":"2026-07-05T02:46:22.210872+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2502.06011","citing_title":"Uncertainty Quantification and Causal Considerations for Off-Policy Decision Making","ref_index":11,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/P6KWOMMXTWA7GLJSV6HT2TJM73","json":"https://pith.science/pith/P6KWOMMXTWA7GLJSV6HT2TJM73.json","graph_json":"https://pith.science/api/pith-number/P6KWOMMXTWA7GLJSV6HT2TJM73/graph.json","events_json":"https://pith.science/api/pith-number/P6KWOMMXTWA7GLJSV6HT2TJM73/events.json","paper":"https://pith.science/paper/P6KWOMMX"},"agent_actions":{"view_html":"https://pith.science/pith/P6KWOMMXTWA7GLJSV6HT2TJM73","download_json":"https://pith.science/pith/P6KWOMMXTWA7GLJSV6HT2TJM73.json","view_paper":"https://pith.science/paper/P6KWOMMX","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=1802.04064&json=true","fetch_graph":"https://pith.science/api/pith-number/P6KWOMMXTWA7GLJSV6HT2TJM73/graph.json","fetch_events":"https://pith.science/api/pith-number/P6KWOMMXTWA7GLJSV6HT2TJM73/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/P6KWOMMXTWA7GLJSV6HT2TJM73/action/timestamp_anchor","attest_storage":"https://pith.science/pith/P6KWOMMXTWA7GLJSV6HT2TJM73/action/storage_attestation","attest_author":"https://pith.science/pith/P6KWOMMXTWA7GLJSV6HT2TJM73/action/author_attestation","sign_citation":"https://pith.science/pith/P6KWOMMXTWA7GLJSV6HT2TJM73/action/citation_signature","submit_replication":"https://pith.science/pith/P6KWOMMXTWA7GLJSV6HT2TJM73/action/replication_record"}},"created_at":"2026-07-05T02:46:22.210872+00:00","updated_at":"2026-07-05T02:46:22.210872+00:00"}