{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2025:WUCPB37L62Q5ZTJHEVNGZSTUFY","short_pith_number":"pith:WUCPB37L","canonical_record":{"source":{"id":"2508.14881","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-08-20T17:54:21Z","cross_cats_sorted":[],"title_canon_sha256":"89aaf228314a743d3f234ff77f495979ea9b807dff84bde2e96f8a44f2cbb5a0","abstract_canon_sha256":"0bc6a0ffb81e5eebe1ed72a381e6975acb6b4133f15469f728b0d6bb2f625a91"},"schema_version":"1.0"},"canonical_sha256":"b504f0efebf6a1dccd27255a6cca742e36b572bfd90ae42794b045a7d126e9aa","source":{"kind":"arxiv","id":"2508.14881","version":2},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2508.14881","created_at":"2026-07-05T11:58:35Z"},{"alias_kind":"arxiv_version","alias_value":"2508.14881v2","created_at":"2026-07-05T11:58:35Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2508.14881","created_at":"2026-07-05T11:58:35Z"},{"alias_kind":"pith_short_12","alias_value":"WUCPB37L62Q5","created_at":"2026-07-05T11:58:35Z"},{"alias_kind":"pith_short_16","alias_value":"WUCPB37L62Q5ZTJH","created_at":"2026-07-05T11:58:35Z"},{"alias_kind":"pith_short_8","alias_value":"WUCPB37L","created_at":"2026-07-05T11:58:35Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2025:WUCPB37L62Q5ZTJHEVNGZSTUFY","target":"record","payload":{"canonical_record":{"source":{"id":"2508.14881","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-08-20T17:54:21Z","cross_cats_sorted":[],"title_canon_sha256":"89aaf228314a743d3f234ff77f495979ea9b807dff84bde2e96f8a44f2cbb5a0","abstract_canon_sha256":"0bc6a0ffb81e5eebe1ed72a381e6975acb6b4133f15469f728b0d6bb2f625a91"},"schema_version":"1.0"},"canonical_sha256":"b504f0efebf6a1dccd27255a6cca742e36b572bfd90ae42794b045a7d126e9aa","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:58:35.787923Z","signature_b64":"dXZmUAEIVoQp4k99eHfgcKiutNFJNI4zw/gHwdpXGH1gwm4ggH9mMkFZ4qMgHImvOc4m+Fu5EZbhd6+o0keYAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b504f0efebf6a1dccd27255a6cca742e36b572bfd90ae42794b045a7d126e9aa","last_reissued_at":"2026-07-05T11:58:35.787434Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:58:35.787434Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2508.14881","source_version":2,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T11:58:35Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"68jTN3Av7m5qEdxmjs6ZLZsEIj1QisxW3YNT61bbDCtlPWziWv/A7rMXhOavmO2LAkkV6r6J1RKJfMTHPpPCDg==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-22T01:24:05.173596Z"},"content_sha256":"52721eb15bf4efe3e2c7ea4a8129421a6982ddb13f0925be044864928e54880a","schema_version":"1.0","event_id":"sha256:52721eb15bf4efe3e2c7ea4a8129421a6982ddb13f0925be044864928e54880a"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2025:WUCPB37L62Q5ZTJHEVNGZSTUFY","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Compute-Optimal Scaling for Value-Based Deep RL","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Aviral Kumar, Michal Nauman, Oleh Rybkin, Pieter Abbeel, Preston Fu, Sergey Levine, Zhiyuan Zhou","submitted_at":"2025-08-20T17:54:21Z","abstract_excerpt":"As models grow larger and training them becomes expensive, it becomes increasingly important to scale training recipes not just to larger models and more data, but to do so in a compute-optimal manner that extracts maximal performance per unit of compute. While such scaling has been well studied for language modeling, reinforcement learning (RL) has received less attention in this regard. In this paper, we investigate compute scaling for online, value-based deep RL. These methods present two primary axes for compute allocation: model capacity and the update-to-data (UTD) ratio. Given a fixed c"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2508.14881","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2508.14881/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T11:58:35Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"56dkKiP7Pt4Qjh5g8GycGfHYqLU+YTdMP31R0sJTBGNxbuBjorgHVjHcHt5M/hllWQNr4gkEDVVSA+XtIRLTBw==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-22T01:24:05.173923Z"},"content_sha256":"f2c394d659e00fad7b3c166da9725ad777f0770837846f60cf107f1cef76586d","schema_version":"1.0","event_id":"sha256:f2c394d659e00fad7b3c166da9725ad777f0770837846f60cf107f1cef76586d"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/WUCPB37L62Q5ZTJHEVNGZSTUFY/bundle.json","state_url":"https://pith.science/pith/WUCPB37L62Q5ZTJHEVNGZSTUFY/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/WUCPB37L62Q5ZTJHEVNGZSTUFY/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-22T01:24:05Z","links":{"resolver":"https://pith.science/pith/WUCPB37L62Q5ZTJHEVNGZSTUFY","bundle":"https://pith.science/pith/WUCPB37L62Q5ZTJHEVNGZSTUFY/bundle.json","state":"https://pith.science/pith/WUCPB37L62Q5ZTJHEVNGZSTUFY/state.json","well_known_bundle":"https://pith.science/.well-known/pith/WUCPB37L62Q5ZTJHEVNGZSTUFY/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2025:WUCPB37L62Q5ZTJHEVNGZSTUFY","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"0bc6a0ffb81e5eebe1ed72a381e6975acb6b4133f15469f728b0d6bb2f625a91","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-08-20T17:54:21Z","title_canon_sha256":"89aaf228314a743d3f234ff77f495979ea9b807dff84bde2e96f8a44f2cbb5a0"},"schema_version":"1.0","source":{"id":"2508.14881","kind":"arxiv","version":2}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2508.14881","created_at":"2026-07-05T11:58:35Z"},{"alias_kind":"arxiv_version","alias_value":"2508.14881v2","created_at":"2026-07-05T11:58:35Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2508.14881","created_at":"2026-07-05T11:58:35Z"},{"alias_kind":"pith_short_12","alias_value":"WUCPB37L62Q5","created_at":"2026-07-05T11:58:35Z"},{"alias_kind":"pith_short_16","alias_value":"WUCPB37L62Q5ZTJH","created_at":"2026-07-05T11:58:35Z"},{"alias_kind":"pith_short_8","alias_value":"WUCPB37L","created_at":"2026-07-05T11:58:35Z"}],"graph_snapshots":[{"event_id":"sha256:f2c394d659e00fad7b3c166da9725ad777f0770837846f60cf107f1cef76586d","target":"graph","created_at":"2026-07-05T11:58:35Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2508.14881/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"As models grow larger and training them becomes expensive, it becomes increasingly important to scale training recipes not just to larger models and more data, but to do so in a compute-optimal manner that extracts maximal performance per unit of compute. While such scaling has been well studied for language modeling, reinforcement learning (RL) has received less attention in this regard. In this paper, we investigate compute scaling for online, value-based deep RL. These methods present two primary axes for compute allocation: model capacity and the update-to-data (UTD) ratio. Given a fixed c","authors_text":"Aviral Kumar, Michal Nauman, Oleh Rybkin, Pieter Abbeel, Preston Fu, Sergey Levine, Zhiyuan Zhou","cross_cats":[],"headline":"","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-08-20T17:54:21Z","title":"Compute-Optimal Scaling for Value-Based Deep RL"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2508.14881","kind":"arxiv","version":2},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:52721eb15bf4efe3e2c7ea4a8129421a6982ddb13f0925be044864928e54880a","target":"record","created_at":"2026-07-05T11:58:35Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"0bc6a0ffb81e5eebe1ed72a381e6975acb6b4133f15469f728b0d6bb2f625a91","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-08-20T17:54:21Z","title_canon_sha256":"89aaf228314a743d3f234ff77f495979ea9b807dff84bde2e96f8a44f2cbb5a0"},"schema_version":"1.0","source":{"id":"2508.14881","kind":"arxiv","version":2}},"canonical_sha256":"b504f0efebf6a1dccd27255a6cca742e36b572bfd90ae42794b045a7d126e9aa","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"b504f0efebf6a1dccd27255a6cca742e36b572bfd90ae42794b045a7d126e9aa","first_computed_at":"2026-07-05T11:58:35.787434Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T11:58:35.787434Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"dXZmUAEIVoQp4k99eHfgcKiutNFJNI4zw/gHwdpXGH1gwm4ggH9mMkFZ4qMgHImvOc4m+Fu5EZbhd6+o0keYAQ==","signature_status":"signed_v1","signed_at":"2026-07-05T11:58:35.787923Z","signed_message":"canonical_sha256_bytes"},"source_id":"2508.14881","source_kind":"arxiv","source_version":2}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:52721eb15bf4efe3e2c7ea4a8129421a6982ddb13f0925be044864928e54880a","sha256:f2c394d659e00fad7b3c166da9725ad777f0770837846f60cf107f1cef76586d"],"state_sha256":"483a09df9a836152e8d257921b4c7fa0636d3ffa8caa51fbfa4529bdec1c766a"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"SOV2Wewabx21TqSa8gqJSV/0FDn+fcecb5A7GtFFuBUcCWBto5z0N+oOHjrIvfZLhnGLJuK0S9i+umHoGLepCg==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-22T01:24:05.177011Z","bundle_sha256":"2ab0387c08d53651bbd90f94672f5d8d1565e0fa1825b199f9125e128d95508f"}}