{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:YOOFY5LSQFAEQLR7HQVZJEU5PQ","short_pith_number":"pith:YOOFY5LS","schema_version":"1.0","canonical_sha256":"c39c5c75728140482e3f3c2b94929d7c2083680d5532ccb38be1ed908c30c2bd","source":{"kind":"arxiv","id":"2410.09754","version":2},"attestation_state":"computed","paper":{"title":"SimBa: Simplicity Bias for Scaling Up Parameters in Deep Reinforcement Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Donghu Kim, Dongyoon Hwang, Hojoon Lee, Hyunseung Kim, Jaegul Choo, Jun Jet Tai, Kaushik Subramanian, Peter R. Wurman, Peter Stone, Takuma Seno","submitted_at":"2024-10-13T07:20:53Z","abstract_excerpt":"Recent advances in CV and NLP have been largely driven by scaling up the number of network parameters, despite traditional theories suggesting that larger networks are prone to overfitting. These large networks avoid overfitting by integrating components that induce a simplicity bias, guiding models toward simple and generalizable solutions. However, in deep RL, designing and scaling up networks have been less explored. Motivated by this opportunity, we present SimBa, an architecture designed to scale up parameters in deep RL by injecting a simplicity bias. SimBa consists of three components: "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.09754","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-10-13T07:20:53Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"3c54e5f19c35ed673a3db1132cea540b3a7d458e04d9ff88725c9b7147252b9c","abstract_canon_sha256":"0dd8e085bff2d8d19833fd44c2c7ede4cc907bfd4b2e4078d8d033d7864cec0a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:11:35.796089Z","signature_b64":"OJK/QLlKXPDOc+hOJw8qV/5DF6C9Y4KGYoDBsDiO1Er9aj/L4ZqLfplTAnl1f9n7Ix6+Hg8aBz9wh/FB1gPgBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"c39c5c75728140482e3f3c2b94929d7c2083680d5532ccb38be1ed908c30c2bd","last_reissued_at":"2026-07-05T11:11:35.795597Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:11:35.795597Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"SimBa: Simplicity Bias for Scaling Up Parameters in Deep Reinforcement Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Donghu Kim, Dongyoon Hwang, Hojoon Lee, Hyunseung Kim, Jaegul Choo, Jun Jet Tai, Kaushik Subramanian, Peter R. Wurman, Peter Stone, Takuma Seno","submitted_at":"2024-10-13T07:20:53Z","abstract_excerpt":"Recent advances in CV and NLP have been largely driven by scaling up the number of network parameters, despite traditional theories suggesting that larger networks are prone to overfitting. These large networks avoid overfitting by integrating components that induce a simplicity bias, guiding models toward simple and generalizable solutions. However, in deep RL, designing and scaling up networks have been less explored. Motivated by this opportunity, we present SimBa, an architecture designed to scale up parameters in deep RL by injecting a simplicity bias. SimBa consists of three components: "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.09754","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.09754/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.09754","created_at":"2026-07-05T11:11:35.795657+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.09754v2","created_at":"2026-07-05T11:11:35.795657+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.09754","created_at":"2026-07-05T11:11:35.795657+00:00"},{"alias_kind":"pith_short_12","alias_value":"YOOFY5LSQFAE","created_at":"2026-07-05T11:11:35.795657+00:00"},{"alias_kind":"pith_short_16","alias_value":"YOOFY5LSQFAEQLR7","created_at":"2026-07-05T11:11:35.795657+00:00"},{"alias_kind":"pith_short_8","alias_value":"YOOFY5LS","created_at":"2026-07-05T11:11:35.795657+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.10236","citing_title":"When Does Non-Uniform Replay Matter in Reinforcement Learning?","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2604.04539","citing_title":"FlashSAC: Fast and Stable Off-Policy Reinforcement Learning for High-Dimensional Robot Control","ref_index":39,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10236","citing_title":"When Does Non-Uniform Replay Matter in Reinforcement Learning?","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11473","citing_title":"TOPPO: Rethinking PPO for Multi-Task Reinforcement Learning with Critic Balancing","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10236","citing_title":"When Does Non-Uniform Replay Matter in Reinforcement Learning?","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2604.04539","citing_title":"FlashSAC: Fast and Stable Off-Policy Reinforcement Learning for High-Dimensional Robot Control","ref_index":39,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/YOOFY5LSQFAEQLR7HQVZJEU5PQ","json":"https://pith.science/pith/YOOFY5LSQFAEQLR7HQVZJEU5PQ.json","graph_json":"https://pith.science/api/pith-number/YOOFY5LSQFAEQLR7HQVZJEU5PQ/graph.json","events_json":"https://pith.science/api/pith-number/YOOFY5LSQFAEQLR7HQVZJEU5PQ/events.json","paper":"https://pith.science/paper/YOOFY5LS"},"agent_actions":{"view_html":"https://pith.science/pith/YOOFY5LSQFAEQLR7HQVZJEU5PQ","download_json":"https://pith.science/pith/YOOFY5LSQFAEQLR7HQVZJEU5PQ.json","view_paper":"https://pith.science/paper/YOOFY5LS","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.09754&json=true","fetch_graph":"https://pith.science/api/pith-number/YOOFY5LSQFAEQLR7HQVZJEU5PQ/graph.json","fetch_events":"https://pith.science/api/pith-number/YOOFY5LSQFAEQLR7HQVZJEU5PQ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/YOOFY5LSQFAEQLR7HQVZJEU5PQ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/YOOFY5LSQFAEQLR7HQVZJEU5PQ/action/storage_attestation","attest_author":"https://pith.science/pith/YOOFY5LSQFAEQLR7HQVZJEU5PQ/action/author_attestation","sign_citation":"https://pith.science/pith/YOOFY5LSQFAEQLR7HQVZJEU5PQ/action/citation_signature","submit_replication":"https://pith.science/pith/YOOFY5LSQFAEQLR7HQVZJEU5PQ/action/replication_record"}},"created_at":"2026-07-05T11:11:35.795657+00:00","updated_at":"2026-07-05T11:11:35.795657+00:00"}