{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:J23SW6VFPEWGW3IGVTQYC4BV76","short_pith_number":"pith:J23SW6VF","schema_version":"1.0","canonical_sha256":"4eb72b7aa5792c6b6d06ace1817035ff9a98867d2080612e365964841c8d9ffe","source":{"kind":"arxiv","id":"2308.15470","version":2},"attestation_state":"computed","paper":{"title":"Policy composition in reinforcement learning via multi-objective policy optimization","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Abbas Abdolmaleki, Ankit Anand, Doina Precup, Jordan Hoffmann, Martin Riedmiller, Nicolas Heess, Shruti Mishra","submitted_at":"2023-08-29T17:50:27Z","abstract_excerpt":"We enable reinforcement learning agents to learn successful behavior policies by utilizing relevant pre-existing teacher policies. The teacher policies are introduced as objectives, in addition to the task objective, in a multi-objective policy optimization setting. Using the Multi-Objective Maximum a Posteriori Policy Optimization algorithm (Abdolmaleki et al. 2020), we show that teacher policies can help speed up learning, particularly in the absence of shaping rewards. In two domains with continuous observation and action spaces, our agents successfully compose teacher policies in sequence "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2308.15470","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2023-08-29T17:50:27Z","cross_cats_sorted":[],"title_canon_sha256":"2c5d7b9ec23f740c7b2baaf145dcebd55c65ca8214805f9d8d7c4f3fa1b16d19","abstract_canon_sha256":"3b6697008e6ebce33ed8a3d1f695052ece3994c5be430d8cdd49d4eb9e46a44f"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:46:08.325985Z","signature_b64":"tW4bSZWoWiKQpXhWhiD0zN19/teO9qzNat3Y31LL9rV6gc6V7L8ckBTroLDbBA1EbddswAqfyZZttJ0izO66Bw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"4eb72b7aa5792c6b6d06ace1817035ff9a98867d2080612e365964841c8d9ffe","last_reissued_at":"2026-07-05T06:46:08.325488Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:46:08.325488Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Policy composition in reinforcement learning via multi-objective policy optimization","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Abbas Abdolmaleki, Ankit Anand, Doina Precup, Jordan Hoffmann, Martin Riedmiller, Nicolas Heess, Shruti Mishra","submitted_at":"2023-08-29T17:50:27Z","abstract_excerpt":"We enable reinforcement learning agents to learn successful behavior policies by utilizing relevant pre-existing teacher policies. The teacher policies are introduced as objectives, in addition to the task objective, in a multi-objective policy optimization setting. Using the Multi-Objective Maximum a Posteriori Policy Optimization algorithm (Abdolmaleki et al. 2020), we show that teacher policies can help speed up learning, particularly in the absence of shaping rewards. In two domains with continuous observation and action spaces, our agents successfully compose teacher policies in sequence "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2308.15470","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2308.15470/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2308.15470","created_at":"2026-07-05T06:46:08.325548+00:00"},{"alias_kind":"arxiv_version","alias_value":"2308.15470v2","created_at":"2026-07-05T06:46:08.325548+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2308.15470","created_at":"2026-07-05T06:46:08.325548+00:00"},{"alias_kind":"pith_short_12","alias_value":"J23SW6VFPEWG","created_at":"2026-07-05T06:46:08.325548+00:00"},{"alias_kind":"pith_short_16","alias_value":"J23SW6VFPEWGW3IG","created_at":"2026-07-05T06:46:08.325548+00:00"},{"alias_kind":"pith_short_8","alias_value":"J23SW6VF","created_at":"2026-07-05T06:46:08.325548+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/J23SW6VFPEWGW3IGVTQYC4BV76","json":"https://pith.science/pith/J23SW6VFPEWGW3IGVTQYC4BV76.json","graph_json":"https://pith.science/api/pith-number/J23SW6VFPEWGW3IGVTQYC4BV76/graph.json","events_json":"https://pith.science/api/pith-number/J23SW6VFPEWGW3IGVTQYC4BV76/events.json","paper":"https://pith.science/paper/J23SW6VF"},"agent_actions":{"view_html":"https://pith.science/pith/J23SW6VFPEWGW3IGVTQYC4BV76","download_json":"https://pith.science/pith/J23SW6VFPEWGW3IGVTQYC4BV76.json","view_paper":"https://pith.science/paper/J23SW6VF","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2308.15470&json=true","fetch_graph":"https://pith.science/api/pith-number/J23SW6VFPEWGW3IGVTQYC4BV76/graph.json","fetch_events":"https://pith.science/api/pith-number/J23SW6VFPEWGW3IGVTQYC4BV76/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/J23SW6VFPEWGW3IGVTQYC4BV76/action/timestamp_anchor","attest_storage":"https://pith.science/pith/J23SW6VFPEWGW3IGVTQYC4BV76/action/storage_attestation","attest_author":"https://pith.science/pith/J23SW6VFPEWGW3IGVTQYC4BV76/action/author_attestation","sign_citation":"https://pith.science/pith/J23SW6VFPEWGW3IGVTQYC4BV76/action/citation_signature","submit_replication":"https://pith.science/pith/J23SW6VFPEWGW3IGVTQYC4BV76/action/replication_record"}},"created_at":"2026-07-05T06:46:08.325548+00:00","updated_at":"2026-07-05T06:46:08.325548+00:00"}