{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2019:IBJRATWG4YSJ7BGOWSKCZ6JS6R","short_pith_number":"pith:IBJRATWG","schema_version":"1.0","canonical_sha256":"4053104ec6e6249f84ceb4942cf932f44c12f6f38a05da3c43f9ea7f01d75405","source":{"kind":"arxiv","id":"1905.13167","version":1},"attestation_state":"computed","paper":{"title":"Defining Admissible Rewards for High Confidence Policy Evaluation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["stat.ML"],"primary_cat":"cs.LG","authors_text":"Barbara E Engelhardt, Finale Doshi-Velez, Niranjani Prasad","submitted_at":"2019-05-30T16:51:49Z","abstract_excerpt":"A key impediment to reinforcement learning (RL) in real applications with limited, batch data is defining a reward function that reflects what we implicitly know about reasonable behaviour for a task and allows for robust off-policy evaluation. In this work, we develop a method to identify an admissible set of reward functions for policies that (a) do not diverge too far from past behaviour, and (b) can be evaluated with high confidence, given only a collection of past trajectories. Together, these ensure that we propose policies that we trust to be implemented in high-risk settings. We demons"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"1905.13167","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2019-05-30T16:51:49Z","cross_cats_sorted":["stat.ML"],"title_canon_sha256":"1481f2b4ae9cb35fd545ffd7b91887ab9b7f2c5ff18ad38af8155abbc007ed11","abstract_canon_sha256":"64e637d01a500d9f55606afba070ff818a8f48d8e4092229d216ecaf4095132c"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-05-17T23:44:38.767986Z","signature_b64":"ayuWl87UetLVr+E2HqpGPmpWEDkaXhUtqeiMtk9esJs/xQTQZQFlLV8ZEbn+vJmThIgG7TW01BS9Kj8qZI1lBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"4053104ec6e6249f84ceb4942cf932f44c12f6f38a05da3c43f9ea7f01d75405","last_reissued_at":"2026-05-17T23:44:38.767557Z","signature_status":"signed_v1","first_computed_at":"2026-05-17T23:44:38.767557Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Defining Admissible Rewards for High Confidence Policy Evaluation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["stat.ML"],"primary_cat":"cs.LG","authors_text":"Barbara E Engelhardt, Finale Doshi-Velez, Niranjani Prasad","submitted_at":"2019-05-30T16:51:49Z","abstract_excerpt":"A key impediment to reinforcement learning (RL) in real applications with limited, batch data is defining a reward function that reflects what we implicitly know about reasonable behaviour for a task and allows for robust off-policy evaluation. In this work, we develop a method to identify an admissible set of reward functions for policies that (a) do not diverge too far from past behaviour, and (b) can be evaluated with high confidence, given only a collection of past trajectories. Together, these ensure that we propose policies that we trust to be implemented in high-risk settings. We demons"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1905.13167","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"1905.13167","created_at":"2026-05-17T23:44:38.767617+00:00"},{"alias_kind":"arxiv_version","alias_value":"1905.13167v1","created_at":"2026-05-17T23:44:38.767617+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1905.13167","created_at":"2026-05-17T23:44:38.767617+00:00"},{"alias_kind":"pith_short_12","alias_value":"IBJRATWG4YSJ","created_at":"2026-05-18T12:33:18.533446+00:00"},{"alias_kind":"pith_short_16","alias_value":"IBJRATWG4YSJ7BGO","created_at":"2026-05-18T12:33:18.533446+00:00"},{"alias_kind":"pith_short_8","alias_value":"IBJRATWG","created_at":"2026-05-18T12:33:18.533446+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/IBJRATWG4YSJ7BGOWSKCZ6JS6R","json":"https://pith.science/pith/IBJRATWG4YSJ7BGOWSKCZ6JS6R.json","graph_json":"https://pith.science/api/pith-number/IBJRATWG4YSJ7BGOWSKCZ6JS6R/graph.json","events_json":"https://pith.science/api/pith-number/IBJRATWG4YSJ7BGOWSKCZ6JS6R/events.json","paper":"https://pith.science/paper/IBJRATWG"},"agent_actions":{"view_html":"https://pith.science/pith/IBJRATWG4YSJ7BGOWSKCZ6JS6R","download_json":"https://pith.science/pith/IBJRATWG4YSJ7BGOWSKCZ6JS6R.json","view_paper":"https://pith.science/paper/IBJRATWG","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=1905.13167&json=true","fetch_graph":"https://pith.science/api/pith-number/IBJRATWG4YSJ7BGOWSKCZ6JS6R/graph.json","fetch_events":"https://pith.science/api/pith-number/IBJRATWG4YSJ7BGOWSKCZ6JS6R/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/IBJRATWG4YSJ7BGOWSKCZ6JS6R/action/timestamp_anchor","attest_storage":"https://pith.science/pith/IBJRATWG4YSJ7BGOWSKCZ6JS6R/action/storage_attestation","attest_author":"https://pith.science/pith/IBJRATWG4YSJ7BGOWSKCZ6JS6R/action/author_attestation","sign_citation":"https://pith.science/pith/IBJRATWG4YSJ7BGOWSKCZ6JS6R/action/citation_signature","submit_replication":"https://pith.science/pith/IBJRATWG4YSJ7BGOWSKCZ6JS6R/action/replication_record"}},"created_at":"2026-05-17T23:44:38.767617+00:00","updated_at":"2026-05-17T23:44:38.767617+00:00"}