{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2019:CPIB64CBARGL3GRDWCXQTSR7OC","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"79539704edf037399c317d65abb77615b9900ed3d665f4290b7b24b37eae26e3","cross_cats_sorted":["stat.ML"],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2019-01-23T22:52:13Z","title_canon_sha256":"9bc2090543a04f812241671728998fcf5dba73e0749f878c4812f644c3ea6758"},"schema_version":"1.0","source":{"id":"1901.08159","kind":"arxiv","version":1}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"1901.08159","created_at":"2026-05-17T23:55:37Z"},{"alias_kind":"arxiv_version","alias_value":"1901.08159v1","created_at":"2026-05-17T23:55:37Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1901.08159","created_at":"2026-05-17T23:55:37Z"},{"alias_kind":"pith_short_12","alias_value":"CPIB64CBARGL","created_at":"2026-05-18T12:33:15Z"},{"alias_kind":"pith_short_16","alias_value":"CPIB64CBARGL3GRD","created_at":"2026-05-18T12:33:15Z"},{"alias_kind":"pith_short_8","alias_value":"CPIB64CB","created_at":"2026-05-18T12:33:15Z"}],"graph_snapshots":[{"event_id":"sha256:e99add5f4ad194ad8d1a86c1b43e73c264078cf8aacabf644c54eda6f3fd45be","target":"graph","created_at":"2026-05-17T23:55:37Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"paper":{"abstract_excerpt":"We describe MELEE, a meta-learning algorithm for learning a good exploration policy in the interactive contextual bandit setting. Here, an algorithm must take actions based on contexts, and learn based only on a reward signal from the action taken, thereby generating an exploration/exploitation trade-off. MELEE addresses this trade-off by learning a good exploration strategy for offline tasks based on synthetic data, on which it can simulate the contextual bandit setting. Based on these simulations, MELEE uses an imitation learning strategy to learn a good exploration policy that can then be a","authors_text":"Amr Sharaf, Hal Daum\\'e III","cross_cats":["stat.ML"],"headline":"","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2019-01-23T22:52:13Z","title":"Meta-Learning for Contextual Bandit Exploration"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1901.08159","kind":"arxiv","version":1},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:97eb3161920940320b4081550b9f84f9cc512fcd3abd8203d56c81b5b4f516cb","target":"record","created_at":"2026-05-17T23:55:37Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"79539704edf037399c317d65abb77615b9900ed3d665f4290b7b24b37eae26e3","cross_cats_sorted":["stat.ML"],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2019-01-23T22:52:13Z","title_canon_sha256":"9bc2090543a04f812241671728998fcf5dba73e0749f878c4812f644c3ea6758"},"schema_version":"1.0","source":{"id":"1901.08159","kind":"arxiv","version":1}},"canonical_sha256":"13d01f7041044cbd9a23b0af09ca3f70a9fd32bae2dee9c996fde4a556509d62","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"13d01f7041044cbd9a23b0af09ca3f70a9fd32bae2dee9c996fde4a556509d62","first_computed_at":"2026-05-17T23:55:37.793261Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-05-17T23:55:37.793261Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"dk7wVI6Ot0wA7jW1rS7EfsEmN3Br0xw5tzyQsTSbj85vGo1SY/hx2dtiJExWBO6vUYEtHKy1kGsIBXqh2RvcBw==","signature_status":"signed_v1","signed_at":"2026-05-17T23:55:37.793761Z","signed_message":"canonical_sha256_bytes"},"source_id":"1901.08159","source_kind":"arxiv","source_version":1}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:97eb3161920940320b4081550b9f84f9cc512fcd3abd8203d56c81b5b4f516cb","sha256:e99add5f4ad194ad8d1a86c1b43e73c264078cf8aacabf644c54eda6f3fd45be"],"state_sha256":"a0daa5ce808bea1b277ffcb06e2ff6c9ff8ae12a538c4cd9c82de7d335a55f88"}