{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:RYK6KQKTIL5ANARV5IWKHDW7GE","short_pith_number":"pith:RYK6KQKT","schema_version":"1.0","canonical_sha256":"8e15e5415342fa068235ea2ca38edf312d80a18922ae2170c46f453fdc8ceec3","source":{"kind":"arxiv","id":"2104.09122","version":1},"attestation_state":"computed","paper":{"title":"Probabilistic Mixture-of-Experts for Efficient Deep Reinforcement Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Hao Dong, Jie Ren, Wei Pan, Yewen Li, Zihan Ding","submitted_at":"2021-04-19T08:21:56Z","abstract_excerpt":"Deep reinforcement learning (DRL) has successfully solved various problems recently, typically with a unimodal policy representation. However, grasping distinguishable skills for some tasks with non-unique optima can be essential for further improving its learning efficiency and performance, which may lead to a multimodal policy represented as a mixture-of-experts (MOE). To our best knowledge, present DRL algorithms for general utility do not deploy this method as policy function approximators due to the potential challenge in its differentiability for policy learning. In this work, we propose"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2104.09122","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2021-04-19T08:21:56Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"7da2d3089bef4c00030b48c3f844770d37f9eefab03a32a1a50edf0c3f890962","abstract_canon_sha256":"0ff0a21cd522f53eb373c4bf088b255ba20729053cf629375ee80733d7227dc9"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T02:33:09.741357Z","signature_b64":"uzFqfzvxV69reUYSfMgPDnKPktXTgK8fqL0HbY/KHCgPLdTtZn9RFspNMrx7btu1GorWQWSd/ZUKsJx79nrODA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"8e15e5415342fa068235ea2ca38edf312d80a18922ae2170c46f453fdc8ceec3","last_reissued_at":"2026-07-05T02:33:09.740757Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T02:33:09.740757Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Probabilistic Mixture-of-Experts for Efficient Deep Reinforcement Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Hao Dong, Jie Ren, Wei Pan, Yewen Li, Zihan Ding","submitted_at":"2021-04-19T08:21:56Z","abstract_excerpt":"Deep reinforcement learning (DRL) has successfully solved various problems recently, typically with a unimodal policy representation. However, grasping distinguishable skills for some tasks with non-unique optima can be essential for further improving its learning efficiency and performance, which may lead to a multimodal policy represented as a mixture-of-experts (MOE). To our best knowledge, present DRL algorithms for general utility do not deploy this method as policy function approximators due to the potential challenge in its differentiability for policy learning. In this work, we propose"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2104.09122","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2104.09122/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2104.09122","created_at":"2026-07-05T02:33:09.740814+00:00"},{"alias_kind":"arxiv_version","alias_value":"2104.09122v1","created_at":"2026-07-05T02:33:09.740814+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2104.09122","created_at":"2026-07-05T02:33:09.740814+00:00"},{"alias_kind":"pith_short_12","alias_value":"RYK6KQKTIL5A","created_at":"2026-07-05T02:33:09.740814+00:00"},{"alias_kind":"pith_short_16","alias_value":"RYK6KQKTIL5ANARV","created_at":"2026-07-05T02:33:09.740814+00:00"},{"alias_kind":"pith_short_8","alias_value":"RYK6KQKT","created_at":"2026-07-05T02:33:09.740814+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.29033","citing_title":"Moment Matching Q-Learning","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2512.08411","citing_title":"Prismatic World Model: Learning Compositional Dynamics for Planning in Hybrid Systems","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09157","citing_title":"Revisiting Mixture Policies in Entropy-Regularized Actor-Critic","ref_index":40,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/RYK6KQKTIL5ANARV5IWKHDW7GE","json":"https://pith.science/pith/RYK6KQKTIL5ANARV5IWKHDW7GE.json","graph_json":"https://pith.science/api/pith-number/RYK6KQKTIL5ANARV5IWKHDW7GE/graph.json","events_json":"https://pith.science/api/pith-number/RYK6KQKTIL5ANARV5IWKHDW7GE/events.json","paper":"https://pith.science/paper/RYK6KQKT"},"agent_actions":{"view_html":"https://pith.science/pith/RYK6KQKTIL5ANARV5IWKHDW7GE","download_json":"https://pith.science/pith/RYK6KQKTIL5ANARV5IWKHDW7GE.json","view_paper":"https://pith.science/paper/RYK6KQKT","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2104.09122&json=true","fetch_graph":"https://pith.science/api/pith-number/RYK6KQKTIL5ANARV5IWKHDW7GE/graph.json","fetch_events":"https://pith.science/api/pith-number/RYK6KQKTIL5ANARV5IWKHDW7GE/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/RYK6KQKTIL5ANARV5IWKHDW7GE/action/timestamp_anchor","attest_storage":"https://pith.science/pith/RYK6KQKTIL5ANARV5IWKHDW7GE/action/storage_attestation","attest_author":"https://pith.science/pith/RYK6KQKTIL5ANARV5IWKHDW7GE/action/author_attestation","sign_citation":"https://pith.science/pith/RYK6KQKTIL5ANARV5IWKHDW7GE/action/citation_signature","submit_replication":"https://pith.science/pith/RYK6KQKTIL5ANARV5IWKHDW7GE/action/replication_record"}},"created_at":"2026-07-05T02:33:09.740814+00:00","updated_at":"2026-07-05T02:33:09.740814+00:00"}