{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2021:CJ3PYQ6I4NVJ763O6K6CSDH2YE","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"1f4fdf95e02de0acff8bdc8fc77aa15b3ef36a03562ebb9c88f0210a26ac17d5","cross_cats_sorted":["cs.LG"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2021-02-08T16:27:09Z","title_canon_sha256":"af02d3618f8c8467223cd84b77a4a7e3637e799aa6d8568d162fcc45e028c14a"},"schema_version":"1.0","source":{"id":"2102.04323","kind":"arxiv","version":2}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2102.04323","created_at":"2026-07-05T03:39:47Z"},{"alias_kind":"arxiv_version","alias_value":"2102.04323v2","created_at":"2026-07-05T03:39:47Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2102.04323","created_at":"2026-07-05T03:39:47Z"},{"alias_kind":"pith_short_12","alias_value":"CJ3PYQ6I4NVJ","created_at":"2026-07-05T03:39:47Z"},{"alias_kind":"pith_short_16","alias_value":"CJ3PYQ6I4NVJ763O","created_at":"2026-07-05T03:39:47Z"},{"alias_kind":"pith_short_8","alias_value":"CJ3PYQ6I","created_at":"2026-07-05T03:39:47Z"}],"graph_snapshots":[{"event_id":"sha256:27f56d809907d30cd7e2a289e9ff432408240ae42e9e1858af62c5c1879c9226","target":"graph","created_at":"2026-07-05T03:39:47Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2102.04323/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"We study the problem of how to construct a set of policies that can be composed together to solve a collection of reinforcement learning tasks. Each task is a different reward function defined as a linear combination of known features. We consider a specific class of policy compositions which we call set improving policies (SIPs): given a set of policies and a set of tasks, a SIP is any composition of the former whose performance is at least as good as that of its constituents across all the tasks. We focus on the most conservative instantiation of SIPs, set-max policies (SMPs), so our analysi","authors_text":"Andre Barreto, Brendan O'Donoghue, Daniel J Mankowitz, Iurii Kemaev, Satinder Singh, Shaobo Hou, Tom Zahavy","cross_cats":["cs.LG"],"headline":"","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2021-02-08T16:27:09Z","title":"Discovering a set of policies for the worst case reward"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2102.04323","kind":"arxiv","version":2},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:96b45edd63457cec324c4258ef2b33686c22324de776f8de1f5b45f88f9cfd66","target":"record","created_at":"2026-07-05T03:39:47Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"1f4fdf95e02de0acff8bdc8fc77aa15b3ef36a03562ebb9c88f0210a26ac17d5","cross_cats_sorted":["cs.LG"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2021-02-08T16:27:09Z","title_canon_sha256":"af02d3618f8c8467223cd84b77a4a7e3637e799aa6d8568d162fcc45e028c14a"},"schema_version":"1.0","source":{"id":"2102.04323","kind":"arxiv","version":2}},"canonical_sha256":"1276fc43c8e36a9ffb6ef2bc290cfac1221ba306939f472b0799749120a6cddd","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"1276fc43c8e36a9ffb6ef2bc290cfac1221ba306939f472b0799749120a6cddd","first_computed_at":"2026-07-05T03:39:47.803289Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T03:39:47.803289Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"RMbj9JFvFo8MwYgAcHvGlYq8irudH/mZfzq21pFGDR71LriCK7H9z4E48bZSBuMsBHpOBPjOFX8ZqndNQ4w7Bg==","signature_status":"signed_v1","signed_at":"2026-07-05T03:39:47.803780Z","signed_message":"canonical_sha256_bytes"},"source_id":"2102.04323","source_kind":"arxiv","source_version":2}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:96b45edd63457cec324c4258ef2b33686c22324de776f8de1f5b45f88f9cfd66","sha256:27f56d809907d30cd7e2a289e9ff432408240ae42e9e1858af62c5c1879c9226"],"state_sha256":"d773ffb18876e5781a164653158a197fc949a977d772a7e2f8f0a3551a56a62f"}