{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2025:XG557ZLOOFXHXLGDEZ4R464PXZ","short_pith_number":"pith:XG557ZLO","canonical_record":{"source":{"id":"2502.15280","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-02-21T08:17:24Z","cross_cats_sorted":[],"title_canon_sha256":"bd19d7af487a450ed1676f53dabad1ff3984cddbf4a613a7cf13787d04f71289","abstract_canon_sha256":"5bbbb5943fb96eece3597a2fecd10051c1c61033cc528ac43fab2afb7b0e9e35"},"schema_version":"1.0"},"canonical_sha256":"b9bbdfe56e716e7bacc326791e7b8fbe77f5fc0b192dae7b0bae8228c8c861b9","source":{"kind":"arxiv","id":"2502.15280","version":2},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2502.15280","created_at":"2026-07-05T11:11:42Z"},{"alias_kind":"arxiv_version","alias_value":"2502.15280v2","created_at":"2026-07-05T11:11:42Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.15280","created_at":"2026-07-05T11:11:42Z"},{"alias_kind":"pith_short_12","alias_value":"XG557ZLOOFXH","created_at":"2026-07-05T11:11:42Z"},{"alias_kind":"pith_short_16","alias_value":"XG557ZLOOFXHXLGD","created_at":"2026-07-05T11:11:42Z"},{"alias_kind":"pith_short_8","alias_value":"XG557ZLO","created_at":"2026-07-05T11:11:42Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2025:XG557ZLOOFXHXLGDEZ4R464PXZ","target":"record","payload":{"canonical_record":{"source":{"id":"2502.15280","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-02-21T08:17:24Z","cross_cats_sorted":[],"title_canon_sha256":"bd19d7af487a450ed1676f53dabad1ff3984cddbf4a613a7cf13787d04f71289","abstract_canon_sha256":"5bbbb5943fb96eece3597a2fecd10051c1c61033cc528ac43fab2afb7b0e9e35"},"schema_version":"1.0"},"canonical_sha256":"b9bbdfe56e716e7bacc326791e7b8fbe77f5fc0b192dae7b0bae8228c8c861b9","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:11:42.201675Z","signature_b64":"kXqXhbfmzLjlMKB5hOzWR4IKU1PAuflCSS00G7doYq7ubPBiIyyiSQVTWfsWZa6UprNY2HOixLvrY60Yrpl4Cg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b9bbdfe56e716e7bacc326791e7b8fbe77f5fc0b192dae7b0bae8228c8c861b9","last_reissued_at":"2026-07-05T11:11:42.201186Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:11:42.201186Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2502.15280","source_version":2,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T11:11:42Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"8NUUMYxW3tEl8sw7Tqt40k6/jyzyfWNHJSWMrA9ws2rgBOeVpNFlL/mtoPJTGJdtfK6I7wzQEl7xAXihMyQXBg==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-07T12:18:22.289118Z"},"content_sha256":"f80a52b600430de631e2578d7219d88dc672c5fe5ddaee8f8468ba9c09689d19","schema_version":"1.0","event_id":"sha256:f80a52b600430de631e2578d7219d88dc672c5fe5ddaee8f8468ba9c09689d19"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2025:XG557ZLOOFXHXLGDEZ4R464PXZ","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Hyperspherical Normalization for Scalable Deep Reinforcement Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Donghu Kim, Hojoon Lee, Jaegul Choo, Peter Stone, Takuma Seno, Youngdo Lee","submitted_at":"2025-02-21T08:17:24Z","abstract_excerpt":"Scaling up the model size and computation has brought consistent performance improvements in supervised learning. However, this lesson often fails to apply to reinforcement learning (RL) because training the model on non-stationary data easily leads to overfitting and unstable optimization. In response, we introduce SimbaV2, a novel RL architecture designed to stabilize optimization by (i) constraining the growth of weight and feature norm by hyperspherical normalization; and (ii) using a distributional value estimation with reward scaling to maintain stable gradients under varying reward magn"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.15280","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.15280/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T11:11:42Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"KCpYLgYGSR111CFMWSXLqqnI3+Rq0X0cO1YsjVK+jeviRTf0fmqUd3O+sPIJPAXDqM7w8ucXMF+f/XEWmaOhAg==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-07T12:18:22.289623Z"},"content_sha256":"948956797c7dc403877f67a7d8f90e50bdf3ccd745662dda518525fc15335697","schema_version":"1.0","event_id":"sha256:948956797c7dc403877f67a7d8f90e50bdf3ccd745662dda518525fc15335697"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/XG557ZLOOFXHXLGDEZ4R464PXZ/bundle.json","state_url":"https://pith.science/pith/XG557ZLOOFXHXLGDEZ4R464PXZ/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/XG557ZLOOFXHXLGDEZ4R464PXZ/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-07T12:18:22Z","links":{"resolver":"https://pith.science/pith/XG557ZLOOFXHXLGDEZ4R464PXZ","bundle":"https://pith.science/pith/XG557ZLOOFXHXLGDEZ4R464PXZ/bundle.json","state":"https://pith.science/pith/XG557ZLOOFXHXLGDEZ4R464PXZ/state.json","well_known_bundle":"https://pith.science/.well-known/pith/XG557ZLOOFXHXLGDEZ4R464PXZ/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2025:XG557ZLOOFXHXLGDEZ4R464PXZ","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"5bbbb5943fb96eece3597a2fecd10051c1c61033cc528ac43fab2afb7b0e9e35","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-02-21T08:17:24Z","title_canon_sha256":"bd19d7af487a450ed1676f53dabad1ff3984cddbf4a613a7cf13787d04f71289"},"schema_version":"1.0","source":{"id":"2502.15280","kind":"arxiv","version":2}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2502.15280","created_at":"2026-07-05T11:11:42Z"},{"alias_kind":"arxiv_version","alias_value":"2502.15280v2","created_at":"2026-07-05T11:11:42Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.15280","created_at":"2026-07-05T11:11:42Z"},{"alias_kind":"pith_short_12","alias_value":"XG557ZLOOFXH","created_at":"2026-07-05T11:11:42Z"},{"alias_kind":"pith_short_16","alias_value":"XG557ZLOOFXHXLGD","created_at":"2026-07-05T11:11:42Z"},{"alias_kind":"pith_short_8","alias_value":"XG557ZLO","created_at":"2026-07-05T11:11:42Z"}],"graph_snapshots":[{"event_id":"sha256:948956797c7dc403877f67a7d8f90e50bdf3ccd745662dda518525fc15335697","target":"graph","created_at":"2026-07-05T11:11:42Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2502.15280/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Scaling up the model size and computation has brought consistent performance improvements in supervised learning. However, this lesson often fails to apply to reinforcement learning (RL) because training the model on non-stationary data easily leads to overfitting and unstable optimization. In response, we introduce SimbaV2, a novel RL architecture designed to stabilize optimization by (i) constraining the growth of weight and feature norm by hyperspherical normalization; and (ii) using a distributional value estimation with reward scaling to maintain stable gradients under varying reward magn","authors_text":"Donghu Kim, Hojoon Lee, Jaegul Choo, Peter Stone, Takuma Seno, Youngdo Lee","cross_cats":[],"headline":"","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-02-21T08:17:24Z","title":"Hyperspherical Normalization for Scalable Deep Reinforcement Learning"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.15280","kind":"arxiv","version":2},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:f80a52b600430de631e2578d7219d88dc672c5fe5ddaee8f8468ba9c09689d19","target":"record","created_at":"2026-07-05T11:11:42Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"5bbbb5943fb96eece3597a2fecd10051c1c61033cc528ac43fab2afb7b0e9e35","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-02-21T08:17:24Z","title_canon_sha256":"bd19d7af487a450ed1676f53dabad1ff3984cddbf4a613a7cf13787d04f71289"},"schema_version":"1.0","source":{"id":"2502.15280","kind":"arxiv","version":2}},"canonical_sha256":"b9bbdfe56e716e7bacc326791e7b8fbe77f5fc0b192dae7b0bae8228c8c861b9","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"b9bbdfe56e716e7bacc326791e7b8fbe77f5fc0b192dae7b0bae8228c8c861b9","first_computed_at":"2026-07-05T11:11:42.201186Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T11:11:42.201186Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"kXqXhbfmzLjlMKB5hOzWR4IKU1PAuflCSS00G7doYq7ubPBiIyyiSQVTWfsWZa6UprNY2HOixLvrY60Yrpl4Cg==","signature_status":"signed_v1","signed_at":"2026-07-05T11:11:42.201675Z","signed_message":"canonical_sha256_bytes"},"source_id":"2502.15280","source_kind":"arxiv","source_version":2}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:f80a52b600430de631e2578d7219d88dc672c5fe5ddaee8f8468ba9c09689d19","sha256:948956797c7dc403877f67a7d8f90e50bdf3ccd745662dda518525fc15335697"],"state_sha256":"a825e54a6ec05773ca707e7d4b5b10fadc91c4842ffc83f9455e83f99e751eab"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"GZzpPimaN2IGHKqrmG1gzILs+uVSZir1PT/utk/xC172OMwEkj/wUiH8Apk1Hj06FOl25DoEHP5nX9Ay858JCg==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-07T12:18:22.293908Z","bundle_sha256":"e913d386192dcbb15c6f15742d57bac25e7c3ead992b98a0b56a368c5cd5014f"}}