{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:XG557ZLOOFXHXLGDEZ4R464PXZ","short_pith_number":"pith:XG557ZLO","schema_version":"1.0","canonical_sha256":"b9bbdfe56e716e7bacc326791e7b8fbe77f5fc0b192dae7b0bae8228c8c861b9","source":{"kind":"arxiv","id":"2502.15280","version":2},"attestation_state":"computed","paper":{"title":"Hyperspherical Normalization for Scalable Deep Reinforcement Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Donghu Kim, Hojoon Lee, Jaegul Choo, Peter Stone, Takuma Seno, Youngdo Lee","submitted_at":"2025-02-21T08:17:24Z","abstract_excerpt":"Scaling up the model size and computation has brought consistent performance improvements in supervised learning. However, this lesson often fails to apply to reinforcement learning (RL) because training the model on non-stationary data easily leads to overfitting and unstable optimization. In response, we introduce SimbaV2, a novel RL architecture designed to stabilize optimization by (i) constraining the growth of weight and feature norm by hyperspherical normalization; and (ii) using a distributional value estimation with reward scaling to maintain stable gradients under varying reward magn"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.15280","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-02-21T08:17:24Z","cross_cats_sorted":[],"title_canon_sha256":"bd19d7af487a450ed1676f53dabad1ff3984cddbf4a613a7cf13787d04f71289","abstract_canon_sha256":"5bbbb5943fb96eece3597a2fecd10051c1c61033cc528ac43fab2afb7b0e9e35"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:11:42.201675Z","signature_b64":"kXqXhbfmzLjlMKB5hOzWR4IKU1PAuflCSS00G7doYq7ubPBiIyyiSQVTWfsWZa6UprNY2HOixLvrY60Yrpl4Cg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b9bbdfe56e716e7bacc326791e7b8fbe77f5fc0b192dae7b0bae8228c8c861b9","last_reissued_at":"2026-07-05T11:11:42.201186Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:11:42.201186Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Hyperspherical Normalization for Scalable Deep Reinforcement Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Donghu Kim, Hojoon Lee, Jaegul Choo, Peter Stone, Takuma Seno, Youngdo Lee","submitted_at":"2025-02-21T08:17:24Z","abstract_excerpt":"Scaling up the model size and computation has brought consistent performance improvements in supervised learning. However, this lesson often fails to apply to reinforcement learning (RL) because training the model on non-stationary data easily leads to overfitting and unstable optimization. In response, we introduce SimbaV2, a novel RL architecture designed to stabilize optimization by (i) constraining the growth of weight and feature norm by hyperspherical normalization; and (ii) using a distributional value estimation with reward scaling to maintain stable gradients under varying reward magn"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.15280","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.15280/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.15280","created_at":"2026-07-05T11:11:42.201239+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.15280v2","created_at":"2026-07-05T11:11:42.201239+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.15280","created_at":"2026-07-05T11:11:42.201239+00:00"},{"alias_kind":"pith_short_12","alias_value":"XG557ZLOOFXH","created_at":"2026-07-05T11:11:42.201239+00:00"},{"alias_kind":"pith_short_16","alias_value":"XG557ZLOOFXHXLGD","created_at":"2026-07-05T11:11:42.201239+00:00"},{"alias_kind":"pith_short_8","alias_value":"XG557ZLO","created_at":"2026-07-05T11:11:42.201239+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":11,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.25971","citing_title":"Improving Neural Network Training by Decoupling the Magnitude and Direction of Weight Vectors","ref_index":179,"is_internal_anchor":false},{"citing_arxiv_id":"2510.02590","citing_title":"Use the Online Network If You Can: Towards Fast and Stable Reinforcement Learning","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10236","citing_title":"When Does Non-Uniform Replay Matter in Reinforcement Learning?","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2605.16692","citing_title":"EfficientTDMPC: Improved MPC Objectives for Sample-Efficient Continuous Control","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2604.04539","citing_title":"FlashSAC: Fast and Stable Off-Policy Reinforcement Learning for High-Dimensional Robot Control","ref_index":40,"is_internal_anchor":false},{"citing_arxiv_id":"2603.04333","citing_title":"What Does Flow Matching Bring To TD Learning?","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10236","citing_title":"When Does Non-Uniform Replay Matter in Reinforcement Learning?","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10236","citing_title":"When Does Non-Uniform Replay Matter in Reinforcement Learning?","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2604.19033","citing_title":"Intentional Updates for Streaming Reinforcement Learning","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2604.04539","citing_title":"FlashSAC: Fast and Stable Off-Policy Reinforcement Learning for High-Dimensional Robot Control","ref_index":40,"is_internal_anchor":false},{"citing_arxiv_id":"2605.04368","citing_title":"Extending Differential Temporal Difference Methods for Episodic Problems","ref_index":3,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/XG557ZLOOFXHXLGDEZ4R464PXZ","json":"https://pith.science/pith/XG557ZLOOFXHXLGDEZ4R464PXZ.json","graph_json":"https://pith.science/api/pith-number/XG557ZLOOFXHXLGDEZ4R464PXZ/graph.json","events_json":"https://pith.science/api/pith-number/XG557ZLOOFXHXLGDEZ4R464PXZ/events.json","paper":"https://pith.science/paper/XG557ZLO"},"agent_actions":{"view_html":"https://pith.science/pith/XG557ZLOOFXHXLGDEZ4R464PXZ","download_json":"https://pith.science/pith/XG557ZLOOFXHXLGDEZ4R464PXZ.json","view_paper":"https://pith.science/paper/XG557ZLO","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.15280&json=true","fetch_graph":"https://pith.science/api/pith-number/XG557ZLOOFXHXLGDEZ4R464PXZ/graph.json","fetch_events":"https://pith.science/api/pith-number/XG557ZLOOFXHXLGDEZ4R464PXZ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/XG557ZLOOFXHXLGDEZ4R464PXZ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/XG557ZLOOFXHXLGDEZ4R464PXZ/action/storage_attestation","attest_author":"https://pith.science/pith/XG557ZLOOFXHXLGDEZ4R464PXZ/action/author_attestation","sign_citation":"https://pith.science/pith/XG557ZLOOFXHXLGDEZ4R464PXZ/action/citation_signature","submit_replication":"https://pith.science/pith/XG557ZLOOFXHXLGDEZ4R464PXZ/action/replication_record"}},"created_at":"2026-07-05T11:11:42.201239+00:00","updated_at":"2026-07-05T11:11:42.201239+00:00"}