{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:DCGAVBLVE36BPIPPBAUKJH57O7","short_pith_number":"pith:DCGAVBLV","schema_version":"1.0","canonical_sha256":"188c0a857526fc17a1ef0828a49fbf77e2452469f8f8a03578115008c6d51cff","source":{"kind":"arxiv","id":"2502.07523","version":2},"attestation_state":"computed","paper":{"title":"Scaling Off-Policy Reinforcement Learning with Batch and Weight Normalization","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Daniel Palenicek, Florian Vogt, Jan Peters, Joe Watson","submitted_at":"2025-02-11T12:55:32Z","abstract_excerpt":"Reinforcement learning has achieved significant milestones, but sample efficiency remains a bottleneck for real-world applications. Recently, CrossQ has demonstrated state-of-the-art sample efficiency with a low update-to-data (UTD) ratio of 1. In this work, we explore CrossQ's scaling behavior with higher UTD ratios. We identify challenges in the training dynamics, which are emphasized by higher UTD ratios. To address these, we integrate weight normalization into the CrossQ framework, a solution that stabilizes training, has been shown to prevent potential loss of plasticity and keeps the eff"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.07523","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-02-11T12:55:32Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"a1c3fe80eb46c3e01a081d8ecf75fa0fa61dc209e457d7cff521d9e019d30427","abstract_canon_sha256":"5cb0a4045b935176844ea5d5ae955c34aadf2ada3efcda141cdbe828dd4ab1a9"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:07:20.314687Z","signature_b64":"p8z8hlDniSvWK0ZfCuHuW54HpAwJBHIXDUb2HhU/g4CNUj3A7uOrFMUvE1/QNtZsnjbrMMkBZUoqesYgXms3CA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"188c0a857526fc17a1ef0828a49fbf77e2452469f8f8a03578115008c6d51cff","last_reissued_at":"2026-07-05T11:07:20.314214Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:07:20.314214Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Scaling Off-Policy Reinforcement Learning with Batch and Weight Normalization","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Daniel Palenicek, Florian Vogt, Jan Peters, Joe Watson","submitted_at":"2025-02-11T12:55:32Z","abstract_excerpt":"Reinforcement learning has achieved significant milestones, but sample efficiency remains a bottleneck for real-world applications. Recently, CrossQ has demonstrated state-of-the-art sample efficiency with a low update-to-data (UTD) ratio of 1. In this work, we explore CrossQ's scaling behavior with higher UTD ratios. We identify challenges in the training dynamics, which are emphasized by higher UTD ratios. To address these, we integrate weight normalization into the CrossQ framework, a solution that stabilizes training, has been shown to prevent potential loss of plasticity and keeps the eff"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.07523","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.07523/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.07523","created_at":"2026-07-05T11:07:20.314264+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.07523v2","created_at":"2026-07-05T11:07:20.314264+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.07523","created_at":"2026-07-05T11:07:20.314264+00:00"},{"alias_kind":"pith_short_12","alias_value":"DCGAVBLVE36B","created_at":"2026-07-05T11:07:20.314264+00:00"},{"alias_kind":"pith_short_16","alias_value":"DCGAVBLVE36BPIPP","created_at":"2026-07-05T11:07:20.314264+00:00"},{"alias_kind":"pith_short_8","alias_value":"DCGAVBLV","created_at":"2026-07-05T11:07:20.314264+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2604.18978","citing_title":"Low-Rank Adaptation for Critic Learning in Off-Policy Reinforcement Learning","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2605.04368","citing_title":"Extending Differential Temporal Difference Methods for Episodic Problems","ref_index":5,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/DCGAVBLVE36BPIPPBAUKJH57O7","json":"https://pith.science/pith/DCGAVBLVE36BPIPPBAUKJH57O7.json","graph_json":"https://pith.science/api/pith-number/DCGAVBLVE36BPIPPBAUKJH57O7/graph.json","events_json":"https://pith.science/api/pith-number/DCGAVBLVE36BPIPPBAUKJH57O7/events.json","paper":"https://pith.science/paper/DCGAVBLV"},"agent_actions":{"view_html":"https://pith.science/pith/DCGAVBLVE36BPIPPBAUKJH57O7","download_json":"https://pith.science/pith/DCGAVBLVE36BPIPPBAUKJH57O7.json","view_paper":"https://pith.science/paper/DCGAVBLV","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.07523&json=true","fetch_graph":"https://pith.science/api/pith-number/DCGAVBLVE36BPIPPBAUKJH57O7/graph.json","fetch_events":"https://pith.science/api/pith-number/DCGAVBLVE36BPIPPBAUKJH57O7/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/DCGAVBLVE36BPIPPBAUKJH57O7/action/timestamp_anchor","attest_storage":"https://pith.science/pith/DCGAVBLVE36BPIPPBAUKJH57O7/action/storage_attestation","attest_author":"https://pith.science/pith/DCGAVBLVE36BPIPPBAUKJH57O7/action/author_attestation","sign_citation":"https://pith.science/pith/DCGAVBLVE36BPIPPBAUKJH57O7/action/citation_signature","submit_replication":"https://pith.science/pith/DCGAVBLVE36BPIPPBAUKJH57O7/action/replication_record"}},"created_at":"2026-07-05T11:07:20.314264+00:00","updated_at":"2026-07-05T11:07:20.314264+00:00"}