{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:ZFY7FMDZWWOVA7FYEHJLCAKSSK","short_pith_number":"pith:ZFY7FMDZ","schema_version":"1.0","canonical_sha256":"c971f2b079b59d507cb821d2b1015292ae821a8f650d2509eb125e6113c8b07e","source":{"kind":"arxiv","id":"2112.15025","version":3},"attestation_state":"computed","paper":{"title":"Constructing a Good Behavior Basis for Transfer using Generalized Policy Updates","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Doina Precup, Safa Alver","submitted_at":"2021-12-30T12:20:46Z","abstract_excerpt":"We study the problem of learning a good set of policies, so that when combined together, they can solve a wide variety of unseen reinforcement learning tasks with no or very little new data. Specifically, we consider the framework of generalized policy evaluation and improvement, in which the rewards for all tasks of interest are assumed to be expressible as a linear combination of a fixed set of features. We show theoretically that, under certain assumptions, having access to a specific set of diverse policies, which we call a set of independent policies, can allow for instantaneously achievi"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2112.15025","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2021-12-30T12:20:46Z","cross_cats_sorted":[],"title_canon_sha256":"566841548c4c54a562d67f9ae429720a916664d9f9c724aef7526f24fa5a1378","abstract_canon_sha256":"10fe5a58928753ba0b2f5d961603e1e8875e600497e20b6bee3ff79a01d48ac3"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T04:05:03.853056Z","signature_b64":"c7dV6jSYMWFwBYnNhyxRwg7WwRnNSMhOIOmIaLa/e8khi1Kin43RE84agXpdYVWVKXZv2km0+/1jf3FLXA+DDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"c971f2b079b59d507cb821d2b1015292ae821a8f650d2509eb125e6113c8b07e","last_reissued_at":"2026-07-05T04:05:03.852648Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T04:05:03.852648Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Constructing a Good Behavior Basis for Transfer using Generalized Policy Updates","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Doina Precup, Safa Alver","submitted_at":"2021-12-30T12:20:46Z","abstract_excerpt":"We study the problem of learning a good set of policies, so that when combined together, they can solve a wide variety of unseen reinforcement learning tasks with no or very little new data. Specifically, we consider the framework of generalized policy evaluation and improvement, in which the rewards for all tasks of interest are assumed to be expressible as a linear combination of a fixed set of features. We show theoretically that, under certain assumptions, having access to a specific set of diverse policies, which we call a set of independent policies, can allow for instantaneously achievi"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2112.15025","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2112.15025/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2112.15025","created_at":"2026-07-05T04:05:03.852702+00:00"},{"alias_kind":"arxiv_version","alias_value":"2112.15025v3","created_at":"2026-07-05T04:05:03.852702+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2112.15025","created_at":"2026-07-05T04:05:03.852702+00:00"},{"alias_kind":"pith_short_12","alias_value":"ZFY7FMDZWWOV","created_at":"2026-07-05T04:05:03.852702+00:00"},{"alias_kind":"pith_short_16","alias_value":"ZFY7FMDZWWOVA7FY","created_at":"2026-07-05T04:05:03.852702+00:00"},{"alias_kind":"pith_short_8","alias_value":"ZFY7FMDZ","created_at":"2026-07-05T04:05:03.852702+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ZFY7FMDZWWOVA7FYEHJLCAKSSK","json":"https://pith.science/pith/ZFY7FMDZWWOVA7FYEHJLCAKSSK.json","graph_json":"https://pith.science/api/pith-number/ZFY7FMDZWWOVA7FYEHJLCAKSSK/graph.json","events_json":"https://pith.science/api/pith-number/ZFY7FMDZWWOVA7FYEHJLCAKSSK/events.json","paper":"https://pith.science/paper/ZFY7FMDZ"},"agent_actions":{"view_html":"https://pith.science/pith/ZFY7FMDZWWOVA7FYEHJLCAKSSK","download_json":"https://pith.science/pith/ZFY7FMDZWWOVA7FYEHJLCAKSSK.json","view_paper":"https://pith.science/paper/ZFY7FMDZ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2112.15025&json=true","fetch_graph":"https://pith.science/api/pith-number/ZFY7FMDZWWOVA7FYEHJLCAKSSK/graph.json","fetch_events":"https://pith.science/api/pith-number/ZFY7FMDZWWOVA7FYEHJLCAKSSK/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ZFY7FMDZWWOVA7FYEHJLCAKSSK/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ZFY7FMDZWWOVA7FYEHJLCAKSSK/action/storage_attestation","attest_author":"https://pith.science/pith/ZFY7FMDZWWOVA7FYEHJLCAKSSK/action/author_attestation","sign_citation":"https://pith.science/pith/ZFY7FMDZWWOVA7FYEHJLCAKSSK/action/citation_signature","submit_replication":"https://pith.science/pith/ZFY7FMDZWWOVA7FYEHJLCAKSSK/action/replication_record"}},"created_at":"2026-07-05T04:05:03.852702+00:00","updated_at":"2026-07-05T04:05:03.852702+00:00"}