{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:SFRCDBGJ6MMAI3VJYWG4M4QII3","short_pith_number":"pith:SFRCDBGJ","schema_version":"1.0","canonical_sha256":"91622184c9f318046ea9c58dc6720846e17e66ecc4864941573f0a9ae429a776","source":{"kind":"arxiv","id":"2112.12320","version":1},"attestation_state":"computed","paper":{"title":"Model Selection in Batch Policy Optimization","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["stat.ML"],"primary_cat":"cs.LG","authors_text":"Bo Dai, George Tucker, Jonathan N. Lee, Ofir Nachum","submitted_at":"2021-12-23T02:31:50Z","abstract_excerpt":"We study the problem of model selection in batch policy optimization: given a fixed, partial-feedback dataset and $M$ model classes, learn a policy with performance that is competitive with the policy derived from the best model class. We formalize the problem in the contextual bandit setting with linear model classes by identifying three sources of error that any model selection algorithm should optimally trade-off in order to be competitive: (1) approximation error, (2) statistical complexity, and (3) coverage. The first two sources are common in model selection for supervised learning, wher"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2112.12320","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2021-12-23T02:31:50Z","cross_cats_sorted":["stat.ML"],"title_canon_sha256":"0ca916b11d268f4ada0161b7075e24130860a94068d5009e9fb7630fa48aca0d","abstract_canon_sha256":"43edf6114855f337a4adfaf6a1a02950fd07a24885a1bd8a39401ad465bb756f"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T03:43:19.084143Z","signature_b64":"9GGarL23apiz7B2jVRphEUf6L2d/rWCmRk2A7/vjurNL83NFlKJ80UzsCiY4stpPVMq+Nd62gEugcuzPL83OCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"91622184c9f318046ea9c58dc6720846e17e66ecc4864941573f0a9ae429a776","last_reissued_at":"2026-07-05T03:43:19.083812Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T03:43:19.083812Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Model Selection in Batch Policy Optimization","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["stat.ML"],"primary_cat":"cs.LG","authors_text":"Bo Dai, George Tucker, Jonathan N. Lee, Ofir Nachum","submitted_at":"2021-12-23T02:31:50Z","abstract_excerpt":"We study the problem of model selection in batch policy optimization: given a fixed, partial-feedback dataset and $M$ model classes, learn a policy with performance that is competitive with the policy derived from the best model class. We formalize the problem in the contextual bandit setting with linear model classes by identifying three sources of error that any model selection algorithm should optimally trade-off in order to be competitive: (1) approximation error, (2) statistical complexity, and (3) coverage. The first two sources are common in model selection for supervised learning, wher"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2112.12320","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2112.12320/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2112.12320","created_at":"2026-07-05T03:43:19.083872+00:00"},{"alias_kind":"arxiv_version","alias_value":"2112.12320v1","created_at":"2026-07-05T03:43:19.083872+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2112.12320","created_at":"2026-07-05T03:43:19.083872+00:00"},{"alias_kind":"pith_short_12","alias_value":"SFRCDBGJ6MMA","created_at":"2026-07-05T03:43:19.083872+00:00"},{"alias_kind":"pith_short_16","alias_value":"SFRCDBGJ6MMAI3VJ","created_at":"2026-07-05T03:43:19.083872+00:00"},{"alias_kind":"pith_short_8","alias_value":"SFRCDBGJ","created_at":"2026-07-05T03:43:19.083872+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/SFRCDBGJ6MMAI3VJYWG4M4QII3","json":"https://pith.science/pith/SFRCDBGJ6MMAI3VJYWG4M4QII3.json","graph_json":"https://pith.science/api/pith-number/SFRCDBGJ6MMAI3VJYWG4M4QII3/graph.json","events_json":"https://pith.science/api/pith-number/SFRCDBGJ6MMAI3VJYWG4M4QII3/events.json","paper":"https://pith.science/paper/SFRCDBGJ"},"agent_actions":{"view_html":"https://pith.science/pith/SFRCDBGJ6MMAI3VJYWG4M4QII3","download_json":"https://pith.science/pith/SFRCDBGJ6MMAI3VJYWG4M4QII3.json","view_paper":"https://pith.science/paper/SFRCDBGJ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2112.12320&json=true","fetch_graph":"https://pith.science/api/pith-number/SFRCDBGJ6MMAI3VJYWG4M4QII3/graph.json","fetch_events":"https://pith.science/api/pith-number/SFRCDBGJ6MMAI3VJYWG4M4QII3/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/SFRCDBGJ6MMAI3VJYWG4M4QII3/action/timestamp_anchor","attest_storage":"https://pith.science/pith/SFRCDBGJ6MMAI3VJYWG4M4QII3/action/storage_attestation","attest_author":"https://pith.science/pith/SFRCDBGJ6MMAI3VJYWG4M4QII3/action/author_attestation","sign_citation":"https://pith.science/pith/SFRCDBGJ6MMAI3VJYWG4M4QII3/action/citation_signature","submit_replication":"https://pith.science/pith/SFRCDBGJ6MMAI3VJYWG4M4QII3/action/replication_record"}},"created_at":"2026-07-05T03:43:19.083872+00:00","updated_at":"2026-07-05T03:43:19.083872+00:00"}