{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2020:RVC7AEKPO6LCPR3MGCRIG32J6J","short_pith_number":"pith:RVC7AEKP","schema_version":"1.0","canonical_sha256":"8d45f0114f779627c76c30a2836f49f269a1b07bfe7eb822b4a1337881ffbb80","source":{"kind":"arxiv","id":"2007.13609","version":1},"attestation_state":"computed","paper":{"title":"Statistical Bootstrapping for Uncertainty Estimation in Off-Policy Evaluation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["stat.ML"],"primary_cat":"cs.LG","authors_text":"Ilya Kostrikov, Ofir Nachum","submitted_at":"2020-07-27T14:49:22Z","abstract_excerpt":"In reinforcement learning, it is typical to use the empirically observed transitions and rewards to estimate the value of a policy via either model-based or Q-fitting approaches. Although straightforward, these techniques in general yield biased estimates of the true value of the policy. In this work, we investigate the potential for statistical bootstrapping to be used as a way to take these biased estimates and produce calibrated confidence intervals for the true value of the policy. We identify conditions - specifically, sufficient data size and sufficient coverage - under which statistical"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2007.13609","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2020-07-27T14:49:22Z","cross_cats_sorted":["stat.ML"],"title_canon_sha256":"47abb67e6ba64c25994893b735cd980214f864035ab23099cd3aafb2d8f595bc","abstract_canon_sha256":"c65fd8ef0ad2b500af2cf280ab6bc8d96429989c410ae8976c6968eabe8663c5"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T01:22:26.034563Z","signature_b64":"bktJ20/n3Yq45g/o8/kjKaC5QTzeNUGM+6EdjHUrtGDBzbUKSxv2RL3eWWtlXixDWQqN7FvUY4e26YVC4PS2DQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"8d45f0114f779627c76c30a2836f49f269a1b07bfe7eb822b4a1337881ffbb80","last_reissued_at":"2026-07-05T01:22:26.034080Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T01:22:26.034080Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Statistical Bootstrapping for Uncertainty Estimation in Off-Policy Evaluation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["stat.ML"],"primary_cat":"cs.LG","authors_text":"Ilya Kostrikov, Ofir Nachum","submitted_at":"2020-07-27T14:49:22Z","abstract_excerpt":"In reinforcement learning, it is typical to use the empirically observed transitions and rewards to estimate the value of a policy via either model-based or Q-fitting approaches. Although straightforward, these techniques in general yield biased estimates of the true value of the policy. In this work, we investigate the potential for statistical bootstrapping to be used as a way to take these biased estimates and produce calibrated confidence intervals for the true value of the policy. We identify conditions - specifically, sufficient data size and sufficient coverage - under which statistical"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2007.13609","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2007.13609/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2007.13609","created_at":"2026-07-05T01:22:26.034135+00:00"},{"alias_kind":"arxiv_version","alias_value":"2007.13609v1","created_at":"2026-07-05T01:22:26.034135+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2007.13609","created_at":"2026-07-05T01:22:26.034135+00:00"},{"alias_kind":"pith_short_12","alias_value":"RVC7AEKPO6LC","created_at":"2026-07-05T01:22:26.034135+00:00"},{"alias_kind":"pith_short_16","alias_value":"RVC7AEKPO6LCPR3M","created_at":"2026-07-05T01:22:26.034135+00:00"},{"alias_kind":"pith_short_8","alias_value":"RVC7AEKP","created_at":"2026-07-05T01:22:26.034135+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.12410","citing_title":"Model-based Bootstrap of Controlled Markov Chains","ref_index":2,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/RVC7AEKPO6LCPR3MGCRIG32J6J","json":"https://pith.science/pith/RVC7AEKPO6LCPR3MGCRIG32J6J.json","graph_json":"https://pith.science/api/pith-number/RVC7AEKPO6LCPR3MGCRIG32J6J/graph.json","events_json":"https://pith.science/api/pith-number/RVC7AEKPO6LCPR3MGCRIG32J6J/events.json","paper":"https://pith.science/paper/RVC7AEKP"},"agent_actions":{"view_html":"https://pith.science/pith/RVC7AEKPO6LCPR3MGCRIG32J6J","download_json":"https://pith.science/pith/RVC7AEKPO6LCPR3MGCRIG32J6J.json","view_paper":"https://pith.science/paper/RVC7AEKP","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2007.13609&json=true","fetch_graph":"https://pith.science/api/pith-number/RVC7AEKPO6LCPR3MGCRIG32J6J/graph.json","fetch_events":"https://pith.science/api/pith-number/RVC7AEKPO6LCPR3MGCRIG32J6J/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/RVC7AEKPO6LCPR3MGCRIG32J6J/action/timestamp_anchor","attest_storage":"https://pith.science/pith/RVC7AEKPO6LCPR3MGCRIG32J6J/action/storage_attestation","attest_author":"https://pith.science/pith/RVC7AEKPO6LCPR3MGCRIG32J6J/action/author_attestation","sign_citation":"https://pith.science/pith/RVC7AEKPO6LCPR3MGCRIG32J6J/action/citation_signature","submit_replication":"https://pith.science/pith/RVC7AEKPO6LCPR3MGCRIG32J6J/action/replication_record"}},"created_at":"2026-07-05T01:22:26.034135+00:00","updated_at":"2026-07-05T01:22:26.034135+00:00"}