{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:CSO363QMMSGATQEAJJSZDO24HY","short_pith_number":"pith:CSO363QM","schema_version":"1.0","canonical_sha256":"149dbf6e0c648c09c0804a6591bb5c3e086c114735138e6d090ef5e5998e3b70","source":{"kind":"arxiv","id":"2210.16859","version":1},"attestation_state":"computed","paper":{"title":"A Solvable Model of Neural Scaling Laws","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["hep-th","stat.ML"],"primary_cat":"cs.LG","authors_text":"Alexander Maloney, Daniel A. Roberts, James Sully","submitted_at":"2022-10-30T15:13:18Z","abstract_excerpt":"Large language models with a huge number of parameters, when trained on near internet-sized number of tokens, have been empirically shown to obey neural scaling laws: specifically, their performance behaves predictably as a power law in either parameters or dataset size until bottlenecked by the other resource. To understand this better, we first identify the necessary properties allowing such scaling laws to arise and then propose a statistical model -- a joint generative data model and random feature model -- that captures this neural scaling phenomenology. By solving this model in the dual "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2210.16859","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2022-10-30T15:13:18Z","cross_cats_sorted":["hep-th","stat.ML"],"title_canon_sha256":"44ae26300d9a9dfad8ee44cedffe96d7b88f90361fae6955dad314c68cc7f41e","abstract_canon_sha256":"aae49049c7c67e01b0f1928567be620103be88639f4828642ab86b47cea82d06"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:11:51.628662Z","signature_b64":"6BZuoPylM2lnfLeXQzPbhfadczvX1pmu6Y6+gntzVMNgCULrVX3HhQ5a5k9kJJgWhBma7N7QthesiCF1xslbBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"149dbf6e0c648c09c0804a6591bb5c3e086c114735138e6d090ef5e5998e3b70","last_reissued_at":"2026-07-05T05:11:51.628236Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:11:51.628236Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"A Solvable Model of Neural Scaling Laws","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["hep-th","stat.ML"],"primary_cat":"cs.LG","authors_text":"Alexander Maloney, Daniel A. Roberts, James Sully","submitted_at":"2022-10-30T15:13:18Z","abstract_excerpt":"Large language models with a huge number of parameters, when trained on near internet-sized number of tokens, have been empirically shown to obey neural scaling laws: specifically, their performance behaves predictably as a power law in either parameters or dataset size until bottlenecked by the other resource. To understand this better, we first identify the necessary properties allowing such scaling laws to arise and then propose a statistical model -- a joint generative data model and random feature model -- that captures this neural scaling phenomenology. By solving this model in the dual "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2210.16859","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2210.16859/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2210.16859","created_at":"2026-07-05T05:11:51.628296+00:00"},{"alias_kind":"arxiv_version","alias_value":"2210.16859v1","created_at":"2026-07-05T05:11:51.628296+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2210.16859","created_at":"2026-07-05T05:11:51.628296+00:00"},{"alias_kind":"pith_short_12","alias_value":"CSO363QMMSGA","created_at":"2026-07-05T05:11:51.628296+00:00"},{"alias_kind":"pith_short_16","alias_value":"CSO363QMMSGATQEA","created_at":"2026-07-05T05:11:51.628296+00:00"},{"alias_kind":"pith_short_8","alias_value":"CSO363QM","created_at":"2026-07-05T05:11:51.628296+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":21,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.25008","citing_title":"Neural Scaling Universality: If Exponents Are Fixed, Time to Understand Coefficients","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2606.26617","citing_title":"Sketched Linear Contrastive Learning: Approximation, Optimization, and Statistical Scaling","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2606.20347","citing_title":"Critical Percolation as a Synthetic Data Model for Interpretability","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2606.18164","citing_title":"Learning Dynamics of Chain-of-Thought State Tracking in a Solvable Transformer Model","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2606.08167","citing_title":"Explaining Data Mixing Scaling Laws","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2605.31244","citing_title":"Spectral Reach: Understanding Neural Scaling as Progress into the Spectral Tail","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2606.28242","citing_title":"How Width and Data Shape Generalization Scaling Laws in Quadratic Neural Networks","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2604.24037","citing_title":"A Limit Theory of Foundation Models: A Mathematical Approach to Understanding Emergent Intelligence and Scaling Laws","ref_index":84,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05076","citing_title":"High-Dimensional Statistics: Reflections on Progress and Open Problems","ref_index":71,"is_internal_anchor":false},{"citing_arxiv_id":"2605.24316","citing_title":"From One-Pass SGD to Data Reuse: Mini-Batch Scaling Laws in Sketched Linear Regression","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2606.29858","citing_title":"Smooth Scaling Laws Hide Stepwise Token Learning","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2605.29548","citing_title":"Why Larger Models Learn More: Effects of Capacity, Interference, and Rare-Task Retention","ref_index":37,"is_internal_anchor":false},{"citing_arxiv_id":"2605.23591","citing_title":"Asymmetric Scaling Laws from Sparse Features","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2605.22341","citing_title":"A Boundary-Layer Mechanism for One-Third Scaling in Online Softmax Classification","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2605.14567","citing_title":"Scaling Laws from Sequential Feature Recovery: A Solvable Hierarchical Model","ref_index":43,"is_internal_anchor":false},{"citing_arxiv_id":"2604.24037","citing_title":"A Limit Theory of Foundation Models: A Mathematical Approach to Understanding Emergent Intelligence and Scaling Laws","ref_index":83,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09189","citing_title":"Practical Scaling Laws: Converting Compute into Performance in a Data-Constrained World","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05683","citing_title":"Spectral Lens: Activation and Gradient Spectra as Diagnostics of LLM Optimization","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06563","citing_title":"Criticality and Saturation in Orthogonal Neural Networks","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05076","citing_title":"High-Dimensional Statistics: Reflections on Progress and Open Problems","ref_index":71,"is_internal_anchor":false},{"citing_arxiv_id":"2505.10465","citing_title":"Superposition Yields Robust Neural Scaling","ref_index":18,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/CSO363QMMSGATQEAJJSZDO24HY","json":"https://pith.science/pith/CSO363QMMSGATQEAJJSZDO24HY.json","graph_json":"https://pith.science/api/pith-number/CSO363QMMSGATQEAJJSZDO24HY/graph.json","events_json":"https://pith.science/api/pith-number/CSO363QMMSGATQEAJJSZDO24HY/events.json","paper":"https://pith.science/paper/CSO363QM"},"agent_actions":{"view_html":"https://pith.science/pith/CSO363QMMSGATQEAJJSZDO24HY","download_json":"https://pith.science/pith/CSO363QMMSGATQEAJJSZDO24HY.json","view_paper":"https://pith.science/paper/CSO363QM","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2210.16859&json=true","fetch_graph":"https://pith.science/api/pith-number/CSO363QMMSGATQEAJJSZDO24HY/graph.json","fetch_events":"https://pith.science/api/pith-number/CSO363QMMSGATQEAJJSZDO24HY/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/CSO363QMMSGATQEAJJSZDO24HY/action/timestamp_anchor","attest_storage":"https://pith.science/pith/CSO363QMMSGATQEAJJSZDO24HY/action/storage_attestation","attest_author":"https://pith.science/pith/CSO363QMMSGATQEAJJSZDO24HY/action/author_attestation","sign_citation":"https://pith.science/pith/CSO363QMMSGATQEAJJSZDO24HY/action/citation_signature","submit_replication":"https://pith.science/pith/CSO363QMMSGATQEAJJSZDO24HY/action/replication_record"}},"created_at":"2026-07-05T05:11:51.628296+00:00","updated_at":"2026-07-05T05:11:51.628296+00:00"}