{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2023:HZKQRGSHEGTMVFRHQXA6357SGC","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"0ccaf5147c337358bbf4870782f6ce5c105ecc6dde93e20df437b608a1e08098","cross_cats_sorted":["cs.AI","cs.SY","eess.SY"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2023-04-05T16:45:11Z","title_canon_sha256":"bc6f1adcc7a4747d9e91ffaf82905688e59041c368fd42722ca896a2ea5f6334"},"schema_version":"1.0","source":{"id":"2304.02574","kind":"arxiv","version":2}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2304.02574","created_at":"2026-07-05T08:38:01Z"},{"alias_kind":"arxiv_version","alias_value":"2304.02574v2","created_at":"2026-07-05T08:38:01Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2304.02574","created_at":"2026-07-05T08:38:01Z"},{"alias_kind":"pith_short_12","alias_value":"HZKQRGSHEGTM","created_at":"2026-07-05T08:38:01Z"},{"alias_kind":"pith_short_16","alias_value":"HZKQRGSHEGTMVFRH","created_at":"2026-07-05T08:38:01Z"},{"alias_kind":"pith_short_8","alias_value":"HZKQRGSH","created_at":"2026-07-05T08:38:01Z"}],"graph_snapshots":[{"event_id":"sha256:d10b800ff3500228b630f7d4dcce4f992df166eb80f904875422d7402dc5dc08","target":"graph","created_at":"2026-07-05T08:38:01Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2304.02574/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Reinforcement Learning aims at identifying and evaluating efficient control policies from data. In many real-world applications, the learner is not allowed to experiment and cannot gather data in an online manner (this is the case when experimenting is expensive, risky or unethical). For such applications, the reward of a given policy (the target policy) must be estimated using historical data gathered under a different policy (the behavior policy). Most methods for this learning task, referred to as Off-Policy Evaluation (OPE), do not come with accuracy and certainty guarantees. We present a ","authors_text":"Alessio Russo, Alexandre Proutiere, Daniele Foffano","cross_cats":["cs.AI","cs.SY","eess.SY"],"headline":"","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2023-04-05T16:45:11Z","title":"Conformal Off-Policy Evaluation in Markov Decision Processes"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2304.02574","kind":"arxiv","version":2},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:feb7c8b941c823d3ad994e2a9e65db1dd18955b7129a44bbdf754dcbabd24af2","target":"record","created_at":"2026-07-05T08:38:01Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"0ccaf5147c337358bbf4870782f6ce5c105ecc6dde93e20df437b608a1e08098","cross_cats_sorted":["cs.AI","cs.SY","eess.SY"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2023-04-05T16:45:11Z","title_canon_sha256":"bc6f1adcc7a4747d9e91ffaf82905688e59041c368fd42722ca896a2ea5f6334"},"schema_version":"1.0","source":{"id":"2304.02574","kind":"arxiv","version":2}},"canonical_sha256":"3e55089a4721a6ca962785c1edf7f2309d2904b3a703b91031dd1e82f53b8ee8","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"3e55089a4721a6ca962785c1edf7f2309d2904b3a703b91031dd1e82f53b8ee8","first_computed_at":"2026-07-05T08:38:01.658108Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T08:38:01.658108Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"FjhtkkVdK9dfTGZRkJnM3snyxnufmVOnjmjiN9K3aflC4BYOayHAGeZ1QP4vAbqkVCRPJqwjwHromTrwO970Ag==","signature_status":"signed_v1","signed_at":"2026-07-05T08:38:01.658621Z","signed_message":"canonical_sha256_bytes"},"source_id":"2304.02574","source_kind":"arxiv","source_version":2}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:feb7c8b941c823d3ad994e2a9e65db1dd18955b7129a44bbdf754dcbabd24af2","sha256:d10b800ff3500228b630f7d4dcce4f992df166eb80f904875422d7402dc5dc08"],"state_sha256":"9c5e4e7c82f7d670190116ff5a5fbd655d4083b985df8d128c67a1dd00b7fd54"}