{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:DMQIA32QYTXDH4MKYXFIQGYPM5","short_pith_number":"pith:DMQIA32Q","schema_version":"1.0","canonical_sha256":"1b20806f50c4ee33f18ac5ca881b0f676c0a002499d982ca14d5a20411013df5","source":{"kind":"arxiv","id":"2310.17146","version":1},"attestation_state":"computed","paper":{"title":"Counterfactual-Augmented Importance Sampling for Semi-Offline Policy Evaluation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Jenna Wiens, Shengpu Tang","submitted_at":"2023-10-26T04:41:19Z","abstract_excerpt":"In applying reinforcement learning (RL) to high-stakes domains, quantitative and qualitative evaluation using observational data can help practitioners understand the generalization performance of new policies. However, this type of off-policy evaluation (OPE) is inherently limited since offline data may not reflect the distribution shifts resulting from the application of new policies. On the other hand, online evaluation by collecting rollouts according to the new policy is often infeasible, as deploying new policies in these domains can be unsafe. In this work, we propose a semi-offline eva"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2310.17146","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2023-10-26T04:41:19Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"21e3bfbcaa16b8258a3f9709f1627b6b4d4bc6c0cfa4ff27516d22fbbe718e6d","abstract_canon_sha256":"65f73d211590664c8519066eddf1ea91e1e96f971e6240360d55fea41376dad2"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:05:20.434056Z","signature_b64":"uNf4yzrHJpzs6IT6v8qgSntlBGQP937TV7q+qwATkdH7WgKEPL8j2T4YUamFoK42I5FoPzHuOODTjeKka41lCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"1b20806f50c4ee33f18ac5ca881b0f676c0a002499d982ca14d5a20411013df5","last_reissued_at":"2026-07-05T07:05:20.433586Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:05:20.433586Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Counterfactual-Augmented Importance Sampling for Semi-Offline Policy Evaluation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Jenna Wiens, Shengpu Tang","submitted_at":"2023-10-26T04:41:19Z","abstract_excerpt":"In applying reinforcement learning (RL) to high-stakes domains, quantitative and qualitative evaluation using observational data can help practitioners understand the generalization performance of new policies. However, this type of off-policy evaluation (OPE) is inherently limited since offline data may not reflect the distribution shifts resulting from the application of new policies. On the other hand, online evaluation by collecting rollouts according to the new policy is often infeasible, as deploying new policies in these domains can be unsafe. In this work, we propose a semi-offline eva"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2310.17146","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2310.17146/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2310.17146","created_at":"2026-07-05T07:05:20.433642+00:00"},{"alias_kind":"arxiv_version","alias_value":"2310.17146v1","created_at":"2026-07-05T07:05:20.433642+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2310.17146","created_at":"2026-07-05T07:05:20.433642+00:00"},{"alias_kind":"pith_short_12","alias_value":"DMQIA32QYTXD","created_at":"2026-07-05T07:05:20.433642+00:00"},{"alias_kind":"pith_short_16","alias_value":"DMQIA32QYTXDH4MK","created_at":"2026-07-05T07:05:20.433642+00:00"},{"alias_kind":"pith_short_8","alias_value":"DMQIA32Q","created_at":"2026-07-05T07:05:20.433642+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2411.19395","citing_title":"Concept-driven Off Policy Evaluation","ref_index":21,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/DMQIA32QYTXDH4MKYXFIQGYPM5","json":"https://pith.science/pith/DMQIA32QYTXDH4MKYXFIQGYPM5.json","graph_json":"https://pith.science/api/pith-number/DMQIA32QYTXDH4MKYXFIQGYPM5/graph.json","events_json":"https://pith.science/api/pith-number/DMQIA32QYTXDH4MKYXFIQGYPM5/events.json","paper":"https://pith.science/paper/DMQIA32Q"},"agent_actions":{"view_html":"https://pith.science/pith/DMQIA32QYTXDH4MKYXFIQGYPM5","download_json":"https://pith.science/pith/DMQIA32QYTXDH4MKYXFIQGYPM5.json","view_paper":"https://pith.science/paper/DMQIA32Q","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2310.17146&json=true","fetch_graph":"https://pith.science/api/pith-number/DMQIA32QYTXDH4MKYXFIQGYPM5/graph.json","fetch_events":"https://pith.science/api/pith-number/DMQIA32QYTXDH4MKYXFIQGYPM5/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/DMQIA32QYTXDH4MKYXFIQGYPM5/action/timestamp_anchor","attest_storage":"https://pith.science/pith/DMQIA32QYTXDH4MKYXFIQGYPM5/action/storage_attestation","attest_author":"https://pith.science/pith/DMQIA32QYTXDH4MKYXFIQGYPM5/action/author_attestation","sign_citation":"https://pith.science/pith/DMQIA32QYTXDH4MKYXFIQGYPM5/action/citation_signature","submit_replication":"https://pith.science/pith/DMQIA32QYTXDH4MKYXFIQGYPM5/action/replication_record"}},"created_at":"2026-07-05T07:05:20.433642+00:00","updated_at":"2026-07-05T07:05:20.433642+00:00"}