{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2026:HRQF7SSZ4MLABR7FK66OPYE23C","short_pith_number":"pith:HRQF7SSZ","schema_version":"1.0","canonical_sha256":"3c605fca59e31600c7e557bce7e09ad8930c2175d7fc4962d116346b0b3743a5","source":{"kind":"arxiv","id":"2607.03385","version":1},"attestation_state":"computed","paper":{"title":"A Hierarchy of Policy Learning Problems","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"stat.ML","authors_text":"Hamsa Bastani, Osbert Bastani, Shihan Chen","submitted_at":"2026-07-03T14:40:47Z","abstract_excerpt":"Policy learning has received substantial attention with the goal of learning policies from observational data for decision-making. A majority of work in this space has focused on developing algorithms for computing policies that minimize regret compared to the optimal policy. However, in many practical settings, there is insufficient data to obtain low regret. As a result, recent work has shifted attention to alternative objectives, most notably, studying whether it is possible to learn an improving policy that statistically significantly outperforms baseline policies. We argue that there is s"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2607.03385","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"stat.ML","submitted_at":"2026-07-03T14:40:47Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"e6792896a4b79293a8c424e3e392a67db8f1b080edd488d9ffa1318c46774528","abstract_canon_sha256":"4dfb63a6a40e879ebd6f7ebe444a17816d74788bfd210ece8096c9584286565a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-07T02:17:40.678627Z","signature_b64":"7qINbpNil5AcJettMsHvjQrxssbefatWfSNnzOOOdAHTeTX/6KpWcdExkuCG1AEtK8iaY4DLoL+eI9dfxnaKBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"3c605fca59e31600c7e557bce7e09ad8930c2175d7fc4962d116346b0b3743a5","last_reissued_at":"2026-07-07T02:17:40.677871Z","signature_status":"signed_v1","first_computed_at":"2026-07-07T02:17:40.677871Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"A Hierarchy of Policy Learning Problems","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"stat.ML","authors_text":"Hamsa Bastani, Osbert Bastani, Shihan Chen","submitted_at":"2026-07-03T14:40:47Z","abstract_excerpt":"Policy learning has received substantial attention with the goal of learning policies from observational data for decision-making. A majority of work in this space has focused on developing algorithms for computing policies that minimize regret compared to the optimal policy. However, in many practical settings, there is insufficient data to obtain low regret. As a result, recent work has shifted attention to alternative objectives, most notably, studying whether it is possible to learn an improving policy that statistically significantly outperforms baseline policies. We argue that there is s"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2607.03385","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2607.03385/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2607.03385","created_at":"2026-07-07T02:17:40.677987+00:00"},{"alias_kind":"arxiv_version","alias_value":"2607.03385v1","created_at":"2026-07-07T02:17:40.677987+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2607.03385","created_at":"2026-07-07T02:17:40.677987+00:00"},{"alias_kind":"pith_short_12","alias_value":"HRQF7SSZ4MLA","created_at":"2026-07-07T02:17:40.677987+00:00"},{"alias_kind":"pith_short_16","alias_value":"HRQF7SSZ4MLABR7F","created_at":"2026-07-07T02:17:40.677987+00:00"},{"alias_kind":"pith_short_8","alias_value":"HRQF7SSZ","created_at":"2026-07-07T02:17:40.677987+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/HRQF7SSZ4MLABR7FK66OPYE23C","json":"https://pith.science/pith/HRQF7SSZ4MLABR7FK66OPYE23C.json","graph_json":"https://pith.science/api/pith-number/HRQF7SSZ4MLABR7FK66OPYE23C/graph.json","events_json":"https://pith.science/api/pith-number/HRQF7SSZ4MLABR7FK66OPYE23C/events.json","paper":"https://pith.science/paper/HRQF7SSZ"},"agent_actions":{"view_html":"https://pith.science/pith/HRQF7SSZ4MLABR7FK66OPYE23C","download_json":"https://pith.science/pith/HRQF7SSZ4MLABR7FK66OPYE23C.json","view_paper":"https://pith.science/paper/HRQF7SSZ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2607.03385&json=true","fetch_graph":"https://pith.science/api/pith-number/HRQF7SSZ4MLABR7FK66OPYE23C/graph.json","fetch_events":"https://pith.science/api/pith-number/HRQF7SSZ4MLABR7FK66OPYE23C/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/HRQF7SSZ4MLABR7FK66OPYE23C/action/timestamp_anchor","attest_storage":"https://pith.science/pith/HRQF7SSZ4MLABR7FK66OPYE23C/action/storage_attestation","attest_author":"https://pith.science/pith/HRQF7SSZ4MLABR7FK66OPYE23C/action/author_attestation","sign_citation":"https://pith.science/pith/HRQF7SSZ4MLABR7FK66OPYE23C/action/citation_signature","submit_replication":"https://pith.science/pith/HRQF7SSZ4MLABR7FK66OPYE23C/action/replication_record"}},"created_at":"2026-07-07T02:17:40.677987+00:00","updated_at":"2026-07-07T02:17:40.677987+00:00"}