{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:SX6S73LOQCZU5WU3H6QV62ZV3N","short_pith_number":"pith:SX6S73LO","schema_version":"1.0","canonical_sha256":"95fd2fed6e80b34eda9b3fa15f6b35db5f4fdd8a262e21b8f3663ea01c7d4d22","source":{"kind":"arxiv","id":"2406.18420","version":1},"attestation_state":"computed","paper":{"title":"Mixture of Experts in a Mixture of RL settings","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Jakob Foerster, Johan Obando-Ceron, Karolina Dziugaite, Pablo Samuel Castro, Timon Willi","submitted_at":"2024-06-26T15:15:15Z","abstract_excerpt":"Mixtures of Experts (MoEs) have gained prominence in (self-)supervised learning due to their enhanced inference efficiency, adaptability to distributed training, and modularity. Previous research has illustrated that MoEs can significantly boost Deep Reinforcement Learning (DRL) performance by expanding the network's parameter count while reducing dormant neurons, thereby enhancing the model's learning capacity and ability to deal with non-stationarity. In this work, we shed more light on MoEs' ability to deal with non-stationarity and investigate MoEs in DRL settings with \"amplified\" non-stat"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2406.18420","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-06-26T15:15:15Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"62031382ee4bd2f5b144284728b0ee90077f99617007e8dc4b33de7c165fd23a","abstract_canon_sha256":"bc5edacf452a1a6688288e39d4127fbcf11b829232cc68ed0d597a959843075e"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:37:08.893302Z","signature_b64":"+LqvsQ5dEikl2vL6ldaN2vSEZgddlDhhtUEFwGU2GG1BDd9nj+X2oexw4klRT5D2GyimbM7la2bZLi7WiuuyDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"95fd2fed6e80b34eda9b3fa15f6b35db5f4fdd8a262e21b8f3663ea01c7d4d22","last_reissued_at":"2026-07-05T08:37:08.892882Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:37:08.892882Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Mixture of Experts in a Mixture of RL settings","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Jakob Foerster, Johan Obando-Ceron, Karolina Dziugaite, Pablo Samuel Castro, Timon Willi","submitted_at":"2024-06-26T15:15:15Z","abstract_excerpt":"Mixtures of Experts (MoEs) have gained prominence in (self-)supervised learning due to their enhanced inference efficiency, adaptability to distributed training, and modularity. Previous research has illustrated that MoEs can significantly boost Deep Reinforcement Learning (DRL) performance by expanding the network's parameter count while reducing dormant neurons, thereby enhancing the model's learning capacity and ability to deal with non-stationarity. In this work, we shed more light on MoEs' ability to deal with non-stationarity and investigate MoEs in DRL settings with \"amplified\" non-stat"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.18420","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.18420/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2406.18420","created_at":"2026-07-05T08:37:08.892947+00:00"},{"alias_kind":"arxiv_version","alias_value":"2406.18420v1","created_at":"2026-07-05T08:37:08.892947+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.18420","created_at":"2026-07-05T08:37:08.892947+00:00"},{"alias_kind":"pith_short_12","alias_value":"SX6S73LOQCZU","created_at":"2026-07-05T08:37:08.892947+00:00"},{"alias_kind":"pith_short_16","alias_value":"SX6S73LOQCZU5WU3","created_at":"2026-07-05T08:37:08.892947+00:00"},{"alias_kind":"pith_short_8","alias_value":"SX6S73LO","created_at":"2026-07-05T08:37:08.892947+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.08116","citing_title":"MORES: Mobile Reasoning-as-a-Service via Distributed LLM Inference-Time Scaling","ref_index":36,"is_internal_anchor":true},{"citing_arxiv_id":"2607.00457","citing_title":"Multi-scale Mixture of World Models for Embodied Agents in Evolving Environments","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2604.09028","citing_title":"Plasticity-Enhanced Multi-Agent Mixture of Experts for Dynamic Objective Adaptation in UAVs-Assisted Emergency Communication Networks","ref_index":35,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/SX6S73LOQCZU5WU3H6QV62ZV3N","json":"https://pith.science/pith/SX6S73LOQCZU5WU3H6QV62ZV3N.json","graph_json":"https://pith.science/api/pith-number/SX6S73LOQCZU5WU3H6QV62ZV3N/graph.json","events_json":"https://pith.science/api/pith-number/SX6S73LOQCZU5WU3H6QV62ZV3N/events.json","paper":"https://pith.science/paper/SX6S73LO"},"agent_actions":{"view_html":"https://pith.science/pith/SX6S73LOQCZU5WU3H6QV62ZV3N","download_json":"https://pith.science/pith/SX6S73LOQCZU5WU3H6QV62ZV3N.json","view_paper":"https://pith.science/paper/SX6S73LO","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2406.18420&json=true","fetch_graph":"https://pith.science/api/pith-number/SX6S73LOQCZU5WU3H6QV62ZV3N/graph.json","fetch_events":"https://pith.science/api/pith-number/SX6S73LOQCZU5WU3H6QV62ZV3N/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/SX6S73LOQCZU5WU3H6QV62ZV3N/action/timestamp_anchor","attest_storage":"https://pith.science/pith/SX6S73LOQCZU5WU3H6QV62ZV3N/action/storage_attestation","attest_author":"https://pith.science/pith/SX6S73LOQCZU5WU3H6QV62ZV3N/action/author_attestation","sign_citation":"https://pith.science/pith/SX6S73LOQCZU5WU3H6QV62ZV3N/action/citation_signature","submit_replication":"https://pith.science/pith/SX6S73LOQCZU5WU3H6QV62ZV3N/action/replication_record"}},"created_at":"2026-07-05T08:37:08.892947+00:00","updated_at":"2026-07-05T08:37:08.892947+00:00"}