{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2017:VNOHCF4MZ3RHJMFBFRJLAH34KI","short_pith_number":"pith:VNOHCF4M","canonical_record":{"source":{"id":"1706.03469","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2017-06-12T05:19:47Z","cross_cats_sorted":[],"title_canon_sha256":"ebcdaf6cf6573ea3529468d0f07dbf84da2524d5a26059d6df5d4356afd47f28","abstract_canon_sha256":"564b9aad40284a15e9357d2ada8081a7a3eb7f980e5aae9b57298c22ecdf9776"},"schema_version":"1.0"},"canonical_sha256":"ab5c71178ccee274b0a12c52b01f7c52245bb41ca22b44e2dde8de360ce215a0","source":{"kind":"arxiv","id":"1706.03469","version":1},"source_aliases":[{"alias_kind":"arxiv","alias_value":"1706.03469","created_at":"2026-05-18T00:42:35Z"},{"alias_kind":"arxiv_version","alias_value":"1706.03469v1","created_at":"2026-05-18T00:42:35Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1706.03469","created_at":"2026-05-18T00:42:35Z"},{"alias_kind":"pith_short_12","alias_value":"VNOHCF4MZ3RH","created_at":"2026-05-18T12:31:49Z"},{"alias_kind":"pith_short_16","alias_value":"VNOHCF4MZ3RHJMFB","created_at":"2026-05-18T12:31:49Z"},{"alias_kind":"pith_short_8","alias_value":"VNOHCF4M","created_at":"2026-05-18T12:31:49Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2017:VNOHCF4MZ3RHJMFBFRJLAH34KI","target":"record","payload":{"canonical_record":{"source":{"id":"1706.03469","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2017-06-12T05:19:47Z","cross_cats_sorted":[],"title_canon_sha256":"ebcdaf6cf6573ea3529468d0f07dbf84da2524d5a26059d6df5d4356afd47f28","abstract_canon_sha256":"564b9aad40284a15e9357d2ada8081a7a3eb7f980e5aae9b57298c22ecdf9776"},"schema_version":"1.0"},"canonical_sha256":"ab5c71178ccee274b0a12c52b01f7c52245bb41ca22b44e2dde8de360ce215a0","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-05-18T00:42:35.629029Z","signature_b64":"eB1pc+ngn1J3qLl1MaHPDFHoYF/H+sr6A+5bqdCI4pRgXjGCsWwSwXwem/hlWR0k+0WKqP0ABy6t0ZW35Ye/Dg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ab5c71178ccee274b0a12c52b01f7c52245bb41ca22b44e2dde8de360ce215a0","last_reissued_at":"2026-05-18T00:42:35.628402Z","signature_status":"signed_v1","first_computed_at":"2026-05-18T00:42:35.628402Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"1706.03469","source_version":1,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-05-18T00:42:35Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"UtQJbXqmRoAdukbue8nQ/0Ll1NRV9IVV+J6a1i0etpujZU1lqzplNjY5BDO9GdeIFATrj8JhmtVxanLd2XhiCg==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-06-05T15:58:43.677831Z"},"content_sha256":"31f5fb56a90e91a373b6623670f723c008284e99f1f24b757be4d0f16b102f33","schema_version":"1.0","event_id":"sha256:31f5fb56a90e91a373b6623670f723c008284e99f1f24b757be4d0f16b102f33"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2017:VNOHCF4MZ3RHJMFBFRJLAH34KI","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Data-Efficient Policy Evaluation Through Behavior Policy Search","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Josiah P. Hanna, Peter Stone, Philip S. Thomas, Scott Niekum","submitted_at":"2017-06-12T05:19:47Z","abstract_excerpt":"We consider the task of evaluating a policy for a Markov decision process (MDP). The standard unbiased technique for evaluating a policy is to deploy the policy and observe its performance. We show that the data collected from deploying a different policy, commonly called the behavior policy, can be used to produce unbiased estimates with lower mean squared error than this standard technique. We derive an analytic expression for the optimal behavior policy --- the behavior policy that minimizes the mean squared error of the resulting estimates. Because this expression depends on terms that are"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1706.03469","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-05-18T00:42:35Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"VdY56CU/wkwEvW52HQD21UiI+pHEwGOXSVgCEUD3qwtFn0zkjIrzLXnsS9BJvrpKL3QOGYUuneWARV9uiqvFAw==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-06-05T15:58:43.678174Z"},"content_sha256":"98514d4f257b0f0e7dbc3256988ec10fb051233a12596c09759b2846e9c3d047","schema_version":"1.0","event_id":"sha256:98514d4f257b0f0e7dbc3256988ec10fb051233a12596c09759b2846e9c3d047"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/VNOHCF4MZ3RHJMFBFRJLAH34KI/bundle.json","state_url":"https://pith.science/pith/VNOHCF4MZ3RHJMFBFRJLAH34KI/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/VNOHCF4MZ3RHJMFBFRJLAH34KI/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-06-05T15:58:43Z","links":{"resolver":"https://pith.science/pith/VNOHCF4MZ3RHJMFBFRJLAH34KI","bundle":"https://pith.science/pith/VNOHCF4MZ3RHJMFBFRJLAH34KI/bundle.json","state":"https://pith.science/pith/VNOHCF4MZ3RHJMFBFRJLAH34KI/state.json","well_known_bundle":"https://pith.science/.well-known/pith/VNOHCF4MZ3RHJMFBFRJLAH34KI/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2017:VNOHCF4MZ3RHJMFBFRJLAH34KI","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"564b9aad40284a15e9357d2ada8081a7a3eb7f980e5aae9b57298c22ecdf9776","cross_cats_sorted":[],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2017-06-12T05:19:47Z","title_canon_sha256":"ebcdaf6cf6573ea3529468d0f07dbf84da2524d5a26059d6df5d4356afd47f28"},"schema_version":"1.0","source":{"id":"1706.03469","kind":"arxiv","version":1}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"1706.03469","created_at":"2026-05-18T00:42:35Z"},{"alias_kind":"arxiv_version","alias_value":"1706.03469v1","created_at":"2026-05-18T00:42:35Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1706.03469","created_at":"2026-05-18T00:42:35Z"},{"alias_kind":"pith_short_12","alias_value":"VNOHCF4MZ3RH","created_at":"2026-05-18T12:31:49Z"},{"alias_kind":"pith_short_16","alias_value":"VNOHCF4MZ3RHJMFB","created_at":"2026-05-18T12:31:49Z"},{"alias_kind":"pith_short_8","alias_value":"VNOHCF4M","created_at":"2026-05-18T12:31:49Z"}],"graph_snapshots":[{"event_id":"sha256:98514d4f257b0f0e7dbc3256988ec10fb051233a12596c09759b2846e9c3d047","target":"graph","created_at":"2026-05-18T00:42:35Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"paper":{"abstract_excerpt":"We consider the task of evaluating a policy for a Markov decision process (MDP). The standard unbiased technique for evaluating a policy is to deploy the policy and observe its performance. We show that the data collected from deploying a different policy, commonly called the behavior policy, can be used to produce unbiased estimates with lower mean squared error than this standard technique. We derive an analytic expression for the optimal behavior policy --- the behavior policy that minimizes the mean squared error of the resulting estimates. Because this expression depends on terms that are","authors_text":"Josiah P. Hanna, Peter Stone, Philip S. Thomas, Scott Niekum","cross_cats":[],"headline":"","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2017-06-12T05:19:47Z","title":"Data-Efficient Policy Evaluation Through Behavior Policy Search"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1706.03469","kind":"arxiv","version":1},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:31f5fb56a90e91a373b6623670f723c008284e99f1f24b757be4d0f16b102f33","target":"record","created_at":"2026-05-18T00:42:35Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"564b9aad40284a15e9357d2ada8081a7a3eb7f980e5aae9b57298c22ecdf9776","cross_cats_sorted":[],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2017-06-12T05:19:47Z","title_canon_sha256":"ebcdaf6cf6573ea3529468d0f07dbf84da2524d5a26059d6df5d4356afd47f28"},"schema_version":"1.0","source":{"id":"1706.03469","kind":"arxiv","version":1}},"canonical_sha256":"ab5c71178ccee274b0a12c52b01f7c52245bb41ca22b44e2dde8de360ce215a0","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"ab5c71178ccee274b0a12c52b01f7c52245bb41ca22b44e2dde8de360ce215a0","first_computed_at":"2026-05-18T00:42:35.628402Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-05-18T00:42:35.628402Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"eB1pc+ngn1J3qLl1MaHPDFHoYF/H+sr6A+5bqdCI4pRgXjGCsWwSwXwem/hlWR0k+0WKqP0ABy6t0ZW35Ye/Dg==","signature_status":"signed_v1","signed_at":"2026-05-18T00:42:35.629029Z","signed_message":"canonical_sha256_bytes"},"source_id":"1706.03469","source_kind":"arxiv","source_version":1}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:31f5fb56a90e91a373b6623670f723c008284e99f1f24b757be4d0f16b102f33","sha256:98514d4f257b0f0e7dbc3256988ec10fb051233a12596c09759b2846e9c3d047"],"state_sha256":"7f34b686d675372638908bdc3ee71bf879fd2ecd15baefbc1512ef9c789b1640"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"thcziwiEotuld9QILdpxAZRZsPJ55PM9DlTO5gTdusuuYlu/m7rWaoDdPNQVY3fVLsUL39fd9AdIWeQkudWhAw==","signed_message":"bundle_sha256_bytes","signed_at":"2026-06-05T15:58:43.680108Z","bundle_sha256":"5218e6f435c11c547024387d77f49b38499a3ee52af5da6541d3d6973406a7d2"}}