{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:2CXMIZ2IW7JYHV5OVHMPRX2TZW","short_pith_number":"pith:2CXMIZ2I","schema_version":"1.0","canonical_sha256":"d0aec46748b7d383d7aea9d8f8df53cd924d3792666ac33c465df3d3bfbed39a","source":{"kind":"arxiv","id":"2506.03758","version":1},"attestation_state":"computed","paper":{"title":"Scaling CrossQ with Weight Normalization","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Daniel Palenicek, Florian Vogt, Jan Peters","submitted_at":"2025-06-04T09:24:17Z","abstract_excerpt":"Reinforcement learning has achieved significant milestones, but sample efficiency remains a bottleneck for real-world applications. Recently, CrossQ has demonstrated state-of-the-art sample efficiency with a low update-to-data (UTD) ratio of 1. In this work, we explore CrossQ's scaling behavior with higher UTD ratios. We identify challenges in the training dynamics which are emphasized by higher UTDs, particularly Q-bias explosion and the growing magnitude of critic network weights. To address this, we integrate weight normalization into the CrossQ framework, a solution that stabilizes trainin"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2506.03758","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-06-04T09:24:17Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"0d6506bba5519dff23a0ea7d99da4b486a76d5cb6f6883c5e7b396f5293f4961","abstract_canon_sha256":"dcb87a15e451da778d2e96f2e4852cbb52c9598f17274e61a837a9a0cfea28b4"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:15:51.474527Z","signature_b64":"ExxmYNVhCj4cehQYMhv2cHA6TuNV7GHqOuSAPKvj/XrhJQj/2EWCA0Nkh22AX53ygYEGxzX8wN/+AHAOunVaBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"d0aec46748b7d383d7aea9d8f8df53cd924d3792666ac33c465df3d3bfbed39a","last_reissued_at":"2026-07-05T11:15:51.474016Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:15:51.474016Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Scaling CrossQ with Weight Normalization","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Daniel Palenicek, Florian Vogt, Jan Peters","submitted_at":"2025-06-04T09:24:17Z","abstract_excerpt":"Reinforcement learning has achieved significant milestones, but sample efficiency remains a bottleneck for real-world applications. Recently, CrossQ has demonstrated state-of-the-art sample efficiency with a low update-to-data (UTD) ratio of 1. In this work, we explore CrossQ's scaling behavior with higher UTD ratios. We identify challenges in the training dynamics which are emphasized by higher UTDs, particularly Q-bias explosion and the growing magnitude of critic network weights. To address this, we integrate weight normalization into the CrossQ framework, a solution that stabilizes trainin"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.03758","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2506.03758/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2506.03758","created_at":"2026-07-05T11:15:51.474074+00:00"},{"alias_kind":"arxiv_version","alias_value":"2506.03758v1","created_at":"2026-07-05T11:15:51.474074+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.03758","created_at":"2026-07-05T11:15:51.474074+00:00"},{"alias_kind":"pith_short_12","alias_value":"2CXMIZ2IW7JY","created_at":"2026-07-05T11:15:51.474074+00:00"},{"alias_kind":"pith_short_16","alias_value":"2CXMIZ2IW7JYHV5O","created_at":"2026-07-05T11:15:51.474074+00:00"},{"alias_kind":"pith_short_8","alias_value":"2CXMIZ2I","created_at":"2026-07-05T11:15:51.474074+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/2CXMIZ2IW7JYHV5OVHMPRX2TZW","json":"https://pith.science/pith/2CXMIZ2IW7JYHV5OVHMPRX2TZW.json","graph_json":"https://pith.science/api/pith-number/2CXMIZ2IW7JYHV5OVHMPRX2TZW/graph.json","events_json":"https://pith.science/api/pith-number/2CXMIZ2IW7JYHV5OVHMPRX2TZW/events.json","paper":"https://pith.science/paper/2CXMIZ2I"},"agent_actions":{"view_html":"https://pith.science/pith/2CXMIZ2IW7JYHV5OVHMPRX2TZW","download_json":"https://pith.science/pith/2CXMIZ2IW7JYHV5OVHMPRX2TZW.json","view_paper":"https://pith.science/paper/2CXMIZ2I","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2506.03758&json=true","fetch_graph":"https://pith.science/api/pith-number/2CXMIZ2IW7JYHV5OVHMPRX2TZW/graph.json","fetch_events":"https://pith.science/api/pith-number/2CXMIZ2IW7JYHV5OVHMPRX2TZW/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/2CXMIZ2IW7JYHV5OVHMPRX2TZW/action/timestamp_anchor","attest_storage":"https://pith.science/pith/2CXMIZ2IW7JYHV5OVHMPRX2TZW/action/storage_attestation","attest_author":"https://pith.science/pith/2CXMIZ2IW7JYHV5OVHMPRX2TZW/action/author_attestation","sign_citation":"https://pith.science/pith/2CXMIZ2IW7JYHV5OVHMPRX2TZW/action/citation_signature","submit_replication":"https://pith.science/pith/2CXMIZ2IW7JYHV5OVHMPRX2TZW/action/replication_record"}},"created_at":"2026-07-05T11:15:51.474074+00:00","updated_at":"2026-07-05T11:15:51.474074+00:00"}