{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2022:CVA4ZYY7JTZGXJQ3FSKGUDB7SK","short_pith_number":"pith:CVA4ZYY7","canonical_record":{"source":{"id":"2206.02000","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2022-06-04T14:32:41Z","cross_cats_sorted":[],"title_canon_sha256":"76a8635f7ce04bc43de51b505ca5441e38dda364910231a17e7d091c8016fb85","abstract_canon_sha256":"0884f63553fd240df027fc52b28d41bf6ccead95bb8caa0170978806ddd182cc"},"schema_version":"1.0"},"canonical_sha256":"1541cce31f4cf26ba61b2c946a0c3f9291215b53dad7c2d5d3e9198869cefa17","source":{"kind":"arxiv","id":"2206.02000","version":1},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2206.02000","created_at":"2026-07-05T04:28:59Z"},{"alias_kind":"arxiv_version","alias_value":"2206.02000v1","created_at":"2026-07-05T04:28:59Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2206.02000","created_at":"2026-07-05T04:28:59Z"},{"alias_kind":"pith_short_12","alias_value":"CVA4ZYY7JTZG","created_at":"2026-07-05T04:28:59Z"},{"alias_kind":"pith_short_16","alias_value":"CVA4ZYY7JTZGXJQ3","created_at":"2026-07-05T04:28:59Z"},{"alias_kind":"pith_short_8","alias_value":"CVA4ZYY7","created_at":"2026-07-05T04:28:59Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2022:CVA4ZYY7JTZGXJQ3FSKGUDB7SK","target":"record","payload":{"canonical_record":{"source":{"id":"2206.02000","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2022-06-04T14:32:41Z","cross_cats_sorted":[],"title_canon_sha256":"76a8635f7ce04bc43de51b505ca5441e38dda364910231a17e7d091c8016fb85","abstract_canon_sha256":"0884f63553fd240df027fc52b28d41bf6ccead95bb8caa0170978806ddd182cc"},"schema_version":"1.0"},"canonical_sha256":"1541cce31f4cf26ba61b2c946a0c3f9291215b53dad7c2d5d3e9198869cefa17","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T04:28:59.286776Z","signature_b64":"UIdNsaPZ/v+aIMJ3pn2fM+sGI742sN5Ptvqf0ashR0NNy12Aj+wfLvC0m5LDyaPblpaqyJMxQMqxnATfTcu/CA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"1541cce31f4cf26ba61b2c946a0c3f9291215b53dad7c2d5d3e9198869cefa17","last_reissued_at":"2026-07-05T04:28:59.286373Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T04:28:59.286373Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2206.02000","source_version":1,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T04:28:59Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"/ch4RwzgUab0MCWFfwVor1D5n9S3wCL1xk27DWxCV/q+6+QzXwDtGgqQZgxqfHau7WQVpT8m8c7Coe09IHNEBg==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-03T18:39:21.167402Z"},"content_sha256":"51634b0e9d7616cc82d1f0c9e6d3e1bc12654edae8c1474e12d61fad4d487b03","schema_version":"1.0","event_id":"sha256:51634b0e9d7616cc82d1f0c9e6d3e1bc12654edae8c1474e12d61fad4d487b03"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2022:CVA4ZYY7JTZGXJQ3FSKGUDB7SK","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Hybrid Value Estimation for Off-policy Evaluation and Offline Reinforcement Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Shengyi Jiang, Xue-Kun Jin, Xu-Hui Liu, Yang Yu","submitted_at":"2022-06-04T14:32:41Z","abstract_excerpt":"Value function estimation is an indispensable subroutine in reinforcement learning, which becomes more challenging in the offline setting. In this paper, we propose Hybrid Value Estimation (HVE) to reduce value estimation error, which trades off bias and variance by balancing between the value estimation from offline data and the learned model. Theoretical analysis discloses that HVE enjoys a better error bound than the direct methods. HVE can be leveraged in both off-policy evaluation and offline reinforcement learning settings. We, therefore, provide two concrete algorithms Off-policy HVE (O"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2206.02000","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2206.02000/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T04:28:59Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"4fT7HWhISLV2Ml859RIIeQNF3YpwJw8LrT6vvFxQ2Qk86MiCv4esGw4E8IM9YEWgv+YwZKAVj6J9uLkghINcCw==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-03T18:39:21.167930Z"},"content_sha256":"08ecf233f1e40abbfdea00be969feb7bf8fc8fa9f50629d05842f8f9230172da","schema_version":"1.0","event_id":"sha256:08ecf233f1e40abbfdea00be969feb7bf8fc8fa9f50629d05842f8f9230172da"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/CVA4ZYY7JTZGXJQ3FSKGUDB7SK/bundle.json","state_url":"https://pith.science/pith/CVA4ZYY7JTZGXJQ3FSKGUDB7SK/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/CVA4ZYY7JTZGXJQ3FSKGUDB7SK/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-03T18:39:21Z","links":{"resolver":"https://pith.science/pith/CVA4ZYY7JTZGXJQ3FSKGUDB7SK","bundle":"https://pith.science/pith/CVA4ZYY7JTZGXJQ3FSKGUDB7SK/bundle.json","state":"https://pith.science/pith/CVA4ZYY7JTZGXJQ3FSKGUDB7SK/state.json","well_known_bundle":"https://pith.science/.well-known/pith/CVA4ZYY7JTZGXJQ3FSKGUDB7SK/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2022:CVA4ZYY7JTZGXJQ3FSKGUDB7SK","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"0884f63553fd240df027fc52b28d41bf6ccead95bb8caa0170978806ddd182cc","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2022-06-04T14:32:41Z","title_canon_sha256":"76a8635f7ce04bc43de51b505ca5441e38dda364910231a17e7d091c8016fb85"},"schema_version":"1.0","source":{"id":"2206.02000","kind":"arxiv","version":1}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2206.02000","created_at":"2026-07-05T04:28:59Z"},{"alias_kind":"arxiv_version","alias_value":"2206.02000v1","created_at":"2026-07-05T04:28:59Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2206.02000","created_at":"2026-07-05T04:28:59Z"},{"alias_kind":"pith_short_12","alias_value":"CVA4ZYY7JTZG","created_at":"2026-07-05T04:28:59Z"},{"alias_kind":"pith_short_16","alias_value":"CVA4ZYY7JTZGXJQ3","created_at":"2026-07-05T04:28:59Z"},{"alias_kind":"pith_short_8","alias_value":"CVA4ZYY7","created_at":"2026-07-05T04:28:59Z"}],"graph_snapshots":[{"event_id":"sha256:08ecf233f1e40abbfdea00be969feb7bf8fc8fa9f50629d05842f8f9230172da","target":"graph","created_at":"2026-07-05T04:28:59Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2206.02000/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Value function estimation is an indispensable subroutine in reinforcement learning, which becomes more challenging in the offline setting. In this paper, we propose Hybrid Value Estimation (HVE) to reduce value estimation error, which trades off bias and variance by balancing between the value estimation from offline data and the learned model. Theoretical analysis discloses that HVE enjoys a better error bound than the direct methods. HVE can be leveraged in both off-policy evaluation and offline reinforcement learning settings. We, therefore, provide two concrete algorithms Off-policy HVE (O","authors_text":"Shengyi Jiang, Xue-Kun Jin, Xu-Hui Liu, Yang Yu","cross_cats":[],"headline":"","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2022-06-04T14:32:41Z","title":"Hybrid Value Estimation for Off-policy Evaluation and Offline Reinforcement Learning"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2206.02000","kind":"arxiv","version":1},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:51634b0e9d7616cc82d1f0c9e6d3e1bc12654edae8c1474e12d61fad4d487b03","target":"record","created_at":"2026-07-05T04:28:59Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"0884f63553fd240df027fc52b28d41bf6ccead95bb8caa0170978806ddd182cc","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2022-06-04T14:32:41Z","title_canon_sha256":"76a8635f7ce04bc43de51b505ca5441e38dda364910231a17e7d091c8016fb85"},"schema_version":"1.0","source":{"id":"2206.02000","kind":"arxiv","version":1}},"canonical_sha256":"1541cce31f4cf26ba61b2c946a0c3f9291215b53dad7c2d5d3e9198869cefa17","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"1541cce31f4cf26ba61b2c946a0c3f9291215b53dad7c2d5d3e9198869cefa17","first_computed_at":"2026-07-05T04:28:59.286373Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T04:28:59.286373Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"UIdNsaPZ/v+aIMJ3pn2fM+sGI742sN5Ptvqf0ashR0NNy12Aj+wfLvC0m5LDyaPblpaqyJMxQMqxnATfTcu/CA==","signature_status":"signed_v1","signed_at":"2026-07-05T04:28:59.286776Z","signed_message":"canonical_sha256_bytes"},"source_id":"2206.02000","source_kind":"arxiv","source_version":1}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:51634b0e9d7616cc82d1f0c9e6d3e1bc12654edae8c1474e12d61fad4d487b03","sha256:08ecf233f1e40abbfdea00be969feb7bf8fc8fa9f50629d05842f8f9230172da"],"state_sha256":"47631dbfe0589f465ff535dea3b1a6df63077c26203f52f51ef04411a116c2fa"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"6AV0Y9GIH/2rDj7BD8+zYz9L76lAKACFja0S2t21pPW1KYeiDtEjV0+DSuKJP6wR7f5JoQgpWfONQ+Ylw6YQAw==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-03T18:39:21.171567Z","bundle_sha256":"f6f2538ccc5a3395da61833db34cff9116d1d57d729c3baad6a58bfd4f7ceab5"}}