{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:E6XGS5BD4O2KBWNNPATFM4QYGB","short_pith_number":"pith:E6XGS5BD","schema_version":"1.0","canonical_sha256":"27ae697423e3b4a0d9ad78265672183050e0add4aab3517dccdcc1bb114adc3b","source":{"kind":"arxiv","id":"2410.01930","version":2},"attestation_state":"computed","paper":{"title":"Don't flatten, tokenize! Unlocking the key to SoftMoE's efficacy in deep RL","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Aaron Courville, Ghada Sokar, Hugo Larochelle, Johan Obando-Ceron, Pablo Samuel Castro","submitted_at":"2024-10-02T18:22:45Z","abstract_excerpt":"The use of deep neural networks in reinforcement learning (RL) often suffers from performance degradation as model size increases. While soft mixtures of experts (SoftMoEs) have recently shown promise in mitigating this issue for online RL, the reasons behind their effectiveness remain largely unknown. In this work we provide an in-depth analysis identifying the key factors driving this performance gain. We discover the surprising result that tokenizing the encoder output, rather than the use of multiple experts, is what is behind the efficacy of SoftMoEs. Indeed, we demonstrate that even with"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.01930","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-10-02T18:22:45Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"0aa33f387694a07abcd91e0308432b5b70082e396809f6026fefbe672250d96a","abstract_canon_sha256":"945d23e7d295dc7671a69e26d952d78c1ed48d6d5723169b9f2324250a1e9f65"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:20:40.233975Z","signature_b64":"0CtDJJqzE4M5VR2fSruxFZaCcDfKq8KSrwirmE/lV0wutP971jrqzQo+Clk5OKWA6dufIfddfLUNBilakS1LAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"27ae697423e3b4a0d9ad78265672183050e0add4aab3517dccdcc1bb114adc3b","last_reissued_at":"2026-07-05T10:20:40.233451Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:20:40.233451Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Don't flatten, tokenize! Unlocking the key to SoftMoE's efficacy in deep RL","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Aaron Courville, Ghada Sokar, Hugo Larochelle, Johan Obando-Ceron, Pablo Samuel Castro","submitted_at":"2024-10-02T18:22:45Z","abstract_excerpt":"The use of deep neural networks in reinforcement learning (RL) often suffers from performance degradation as model size increases. While soft mixtures of experts (SoftMoEs) have recently shown promise in mitigating this issue for online RL, the reasons behind their effectiveness remain largely unknown. In this work we provide an in-depth analysis identifying the key factors driving this performance gain. We discover the surprising result that tokenizing the encoder output, rather than the use of multiple experts, is what is behind the efficacy of SoftMoEs. Indeed, we demonstrate that even with"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.01930","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.01930/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.01930","created_at":"2026-07-05T10:20:40.233512+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.01930v2","created_at":"2026-07-05T10:20:40.233512+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.01930","created_at":"2026-07-05T10:20:40.233512+00:00"},{"alias_kind":"pith_short_12","alias_value":"E6XGS5BD4O2K","created_at":"2026-07-05T10:20:40.233512+00:00"},{"alias_kind":"pith_short_16","alias_value":"E6XGS5BD4O2KBWNN","created_at":"2026-07-05T10:20:40.233512+00:00"},{"alias_kind":"pith_short_8","alias_value":"E6XGS5BD","created_at":"2026-07-05T10:20:40.233512+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2506.13672","citing_title":"The Courage to Stop: Overcoming Sunk Cost Fallacy in Deep Reinforcement Learning","ref_index":52,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/E6XGS5BD4O2KBWNNPATFM4QYGB","json":"https://pith.science/pith/E6XGS5BD4O2KBWNNPATFM4QYGB.json","graph_json":"https://pith.science/api/pith-number/E6XGS5BD4O2KBWNNPATFM4QYGB/graph.json","events_json":"https://pith.science/api/pith-number/E6XGS5BD4O2KBWNNPATFM4QYGB/events.json","paper":"https://pith.science/paper/E6XGS5BD"},"agent_actions":{"view_html":"https://pith.science/pith/E6XGS5BD4O2KBWNNPATFM4QYGB","download_json":"https://pith.science/pith/E6XGS5BD4O2KBWNNPATFM4QYGB.json","view_paper":"https://pith.science/paper/E6XGS5BD","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.01930&json=true","fetch_graph":"https://pith.science/api/pith-number/E6XGS5BD4O2KBWNNPATFM4QYGB/graph.json","fetch_events":"https://pith.science/api/pith-number/E6XGS5BD4O2KBWNNPATFM4QYGB/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/E6XGS5BD4O2KBWNNPATFM4QYGB/action/timestamp_anchor","attest_storage":"https://pith.science/pith/E6XGS5BD4O2KBWNNPATFM4QYGB/action/storage_attestation","attest_author":"https://pith.science/pith/E6XGS5BD4O2KBWNNPATFM4QYGB/action/author_attestation","sign_citation":"https://pith.science/pith/E6XGS5BD4O2KBWNNPATFM4QYGB/action/citation_signature","submit_replication":"https://pith.science/pith/E6XGS5BD4O2KBWNNPATFM4QYGB/action/replication_record"}},"created_at":"2026-07-05T10:20:40.233512+00:00","updated_at":"2026-07-05T10:20:40.233512+00:00"}