{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:RKC6QSF3G3YDHXMWCREYPGPODK","short_pith_number":"pith:RKC6QSF3","schema_version":"1.0","canonical_sha256":"8a85e848bb36f033dd9614498799ee1aaaa3bd411653f2216b0b618befe64b29","source":{"kind":"arxiv","id":"2507.02119","version":2},"attestation_state":"computed","paper":{"title":"Scaling Collapse Reveals Universal Dynamics in Compute-Optimally Trained Neural Networks","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Andrew Gordon Wilson, Atish Agarwala, Jeffrey Pennington, Lechao Xiao, Shikai Qiu","submitted_at":"2025-07-02T20:03:34Z","abstract_excerpt":"What scaling limits govern neural network training dynamics when model size and training time grow in tandem? We show that despite the complex interactions between architecture, training algorithms, and data, compute-optimally trained models exhibit a remarkably precise universality. Specifically, loss curves from models of varying sizes collapse onto a single universal curve when training compute and loss are normalized to unity at the end of training. With learning rate decay, the collapse becomes so tight that differences in the normalized curves across models fall below the noise floor of "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2507.02119","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2025-07-02T20:03:34Z","cross_cats_sorted":[],"title_canon_sha256":"c52bfb5bcf85a01b7e614e01b64de1cf9fb8ae907c3f91d80af27507ccbbb5b4","abstract_canon_sha256":"fed41e9876400418bc1ad921db6bb1d8ef63d7b044164e711270fdc83f016558"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:32:33.764156Z","signature_b64":"qOFCFvoDKK91p/Ft1yJg4nD+TGudJsxkM9jYUni/ohrdJm8Iqk88iwLqwm3+teT7ifxmu34dyooR2pF/o10rDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"8a85e848bb36f033dd9614498799ee1aaaa3bd411653f2216b0b618befe64b29","last_reissued_at":"2026-07-05T11:32:33.763650Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:32:33.763650Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Scaling Collapse Reveals Universal Dynamics in Compute-Optimally Trained Neural Networks","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Andrew Gordon Wilson, Atish Agarwala, Jeffrey Pennington, Lechao Xiao, Shikai Qiu","submitted_at":"2025-07-02T20:03:34Z","abstract_excerpt":"What scaling limits govern neural network training dynamics when model size and training time grow in tandem? We show that despite the complex interactions between architecture, training algorithms, and data, compute-optimally trained models exhibit a remarkably precise universality. Specifically, loss curves from models of varying sizes collapse onto a single universal curve when training compute and loss are normalized to unity at the end of training. With learning rate decay, the collapse becomes so tight that differences in the normalized curves across models fall below the noise floor of "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2507.02119","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2507.02119/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2507.02119","created_at":"2026-07-05T11:32:33.763711+00:00"},{"alias_kind":"arxiv_version","alias_value":"2507.02119v2","created_at":"2026-07-05T11:32:33.763711+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2507.02119","created_at":"2026-07-05T11:32:33.763711+00:00"},{"alias_kind":"pith_short_12","alias_value":"RKC6QSF3G3YD","created_at":"2026-07-05T11:32:33.763711+00:00"},{"alias_kind":"pith_short_16","alias_value":"RKC6QSF3G3YDHXMW","created_at":"2026-07-05T11:32:33.763711+00:00"},{"alias_kind":"pith_short_8","alias_value":"RKC6QSF3","created_at":"2026-07-05T11:32:33.763711+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2602.04774","citing_title":"Theory of Optimal Learning Rate Schedules and Scaling Laws for a Random Feature Model","ref_index":18,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/RKC6QSF3G3YDHXMWCREYPGPODK","json":"https://pith.science/pith/RKC6QSF3G3YDHXMWCREYPGPODK.json","graph_json":"https://pith.science/api/pith-number/RKC6QSF3G3YDHXMWCREYPGPODK/graph.json","events_json":"https://pith.science/api/pith-number/RKC6QSF3G3YDHXMWCREYPGPODK/events.json","paper":"https://pith.science/paper/RKC6QSF3"},"agent_actions":{"view_html":"https://pith.science/pith/RKC6QSF3G3YDHXMWCREYPGPODK","download_json":"https://pith.science/pith/RKC6QSF3G3YDHXMWCREYPGPODK.json","view_paper":"https://pith.science/paper/RKC6QSF3","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2507.02119&json=true","fetch_graph":"https://pith.science/api/pith-number/RKC6QSF3G3YDHXMWCREYPGPODK/graph.json","fetch_events":"https://pith.science/api/pith-number/RKC6QSF3G3YDHXMWCREYPGPODK/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/RKC6QSF3G3YDHXMWCREYPGPODK/action/timestamp_anchor","attest_storage":"https://pith.science/pith/RKC6QSF3G3YDHXMWCREYPGPODK/action/storage_attestation","attest_author":"https://pith.science/pith/RKC6QSF3G3YDHXMWCREYPGPODK/action/author_attestation","sign_citation":"https://pith.science/pith/RKC6QSF3G3YDHXMWCREYPGPODK/action/citation_signature","submit_replication":"https://pith.science/pith/RKC6QSF3G3YDHXMWCREYPGPODK/action/replication_record"}},"created_at":"2026-07-05T11:32:33.763711+00:00","updated_at":"2026-07-05T11:32:33.763711+00:00"}