{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2026:WQEG76N3JINWH4UCPV3IWWYKFF","short_pith_number":"pith:WQEG76N3","schema_version":"1.0","canonical_sha256":"b4086ff9bb4a1b63f2827d768b5b0a297d93331ada2521ee7371f2ab6b5277de","source":{"kind":"arxiv","id":"2602.07488","version":3},"attestation_state":"computed","paper":{"title":"Deriving Neural Scaling Laws from the statistics of natural language","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","stat.ML"],"primary_cat":"cs.LG","authors_text":"Allan Ravent\\'os, Francesco Cagnetta, Matthieu Wyart, Surya Ganguli","submitted_at":"2026-02-07T10:40:28Z","abstract_excerpt":"Despite the fact that experimental neural scaling laws have substantially guided empirical progress in large-scale machine learning, no existing theory can quantitatively predict the exponents of these important laws for any modern LLM trained on any natural language dataset. We provide the first such theory in the case of data-limited scaling laws. We isolate two key statistical properties of language that alone can predict neural scaling exponents: (i) the decay of pairwise token correlations with time separation between token pairs, and (ii) the decay of the next-token conditional entropy w"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2602.07488","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2026-02-07T10:40:28Z","cross_cats_sorted":["cs.AI","stat.ML"],"title_canon_sha256":"3e58a8beca57b64ee938cfbd1abd53d5f48cc323e3dc6b8e76780a7a01792bb7","abstract_canon_sha256":"e3f0351cde33aa708490a75b27b4cd79b43a3065d1c072ca8a9546c3025c4c13"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-07T00:15:52.253113Z","signature_b64":"eFr1m6mr6rOzVnWumzoA4L08LvR0YssCOJd9RHJriLgDQs4nDSKLDw51vC2W15M633azojFOoUU30lIapdDRDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b4086ff9bb4a1b63f2827d768b5b0a297d93331ada2521ee7371f2ab6b5277de","last_reissued_at":"2026-07-07T00:15:52.251830Z","signature_status":"signed_v1","first_computed_at":"2026-07-07T00:15:52.251830Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Deriving Neural Scaling Laws from the statistics of natural language","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","stat.ML"],"primary_cat":"cs.LG","authors_text":"Allan Ravent\\'os, Francesco Cagnetta, Matthieu Wyart, Surya Ganguli","submitted_at":"2026-02-07T10:40:28Z","abstract_excerpt":"Despite the fact that experimental neural scaling laws have substantially guided empirical progress in large-scale machine learning, no existing theory can quantitatively predict the exponents of these important laws for any modern LLM trained on any natural language dataset. We provide the first such theory in the case of data-limited scaling laws. We isolate two key statistical properties of language that alone can predict neural scaling exponents: (i) the decay of pairwise token correlations with time separation between token pairs, and (ii) the decay of the next-token conditional entropy w"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2602.07488","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2602.07488/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2602.07488","created_at":"2026-07-07T00:15:52.252290+00:00"},{"alias_kind":"arxiv_version","alias_value":"2602.07488v3","created_at":"2026-07-07T00:15:52.252290+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2602.07488","created_at":"2026-07-07T00:15:52.252290+00:00"},{"alias_kind":"pith_short_12","alias_value":"WQEG76N3JINW","created_at":"2026-07-07T00:15:52.252290+00:00"},{"alias_kind":"pith_short_16","alias_value":"WQEG76N3JINWH4UC","created_at":"2026-07-07T00:15:52.252290+00:00"},{"alias_kind":"pith_short_8","alias_value":"WQEG76N3","created_at":"2026-07-07T00:15:52.252290+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":16,"internal_anchor_count":16,"sample":[{"citing_arxiv_id":"2606.25008","citing_title":"Neural Scaling Universality: If Exponents Are Fixed, Time to Understand Coefficients","ref_index":42,"is_internal_anchor":true},{"citing_arxiv_id":"2606.20347","citing_title":"Critical Percolation as a Synthetic Data Model for Interpretability","ref_index":13,"is_internal_anchor":true},{"citing_arxiv_id":"2606.20299","citing_title":"Statistical Properties of Training & Generalization","ref_index":155,"is_internal_anchor":true},{"citing_arxiv_id":"2606.03990","citing_title":"Neuron Populations Exhibit Divergent Selectivity with Scale","ref_index":3,"is_internal_anchor":true},{"citing_arxiv_id":"2606.02794","citing_title":"Scaling Laws for Neural-Network Quantum States","ref_index":9,"is_internal_anchor":true},{"citing_arxiv_id":"2606.28103","citing_title":"Phase structure of the Random Language Model","ref_index":13,"is_internal_anchor":true},{"citing_arxiv_id":"2606.28242","citing_title":"How Width and Data Shape Generalization Scaling Laws in Quadratic Neural Networks","ref_index":7,"is_internal_anchor":true},{"citing_arxiv_id":"2606.29858","citing_title":"Smooth Scaling Laws Hide Stepwise Token Learning","ref_index":40,"is_internal_anchor":true},{"citing_arxiv_id":"2605.27006","citing_title":"Sampling Data with Chains of Forward-Backward Diffusion Steps","ref_index":31,"is_internal_anchor":true},{"citing_arxiv_id":"2605.27734","citing_title":"Learn from your own latents and not from tokens: A sample-complexity theory","ref_index":34,"is_internal_anchor":true},{"citing_arxiv_id":"2605.29548","citing_title":"Why Larger Models Learn More: Effects of Capacity, Interference, and Rare-Task Retention","ref_index":41,"is_internal_anchor":true},{"citing_arxiv_id":"2606.20299","citing_title":"Statistical Properties of Training & Generalization","ref_index":155,"is_internal_anchor":true},{"citing_arxiv_id":"2605.23591","citing_title":"Asymmetric Scaling Laws from Sparse Features","ref_index":8,"is_internal_anchor":true},{"citing_arxiv_id":"2605.13612","citing_title":"Deep Learning as Neural Low-Degree Filtering: A Spectral Theory of Hierarchical Feature Learning","ref_index":47,"is_internal_anchor":true},{"citing_arxiv_id":"2604.26841","citing_title":"Language Diffusion Models are Associative Memories Capable of Retrieving Unseen Data","ref_index":43,"is_internal_anchor":true},{"citing_arxiv_id":"2604.21691","citing_title":"There Will Be a Scientific Theory of Deep Learning","ref_index":254,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/WQEG76N3JINWH4UCPV3IWWYKFF","json":"https://pith.science/pith/WQEG76N3JINWH4UCPV3IWWYKFF.json","graph_json":"https://pith.science/api/pith-number/WQEG76N3JINWH4UCPV3IWWYKFF/graph.json","events_json":"https://pith.science/api/pith-number/WQEG76N3JINWH4UCPV3IWWYKFF/events.json","paper":"https://pith.science/paper/WQEG76N3"},"agent_actions":{"view_html":"https://pith.science/pith/WQEG76N3JINWH4UCPV3IWWYKFF","download_json":"https://pith.science/pith/WQEG76N3JINWH4UCPV3IWWYKFF.json","view_paper":"https://pith.science/paper/WQEG76N3","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2602.07488&json=true","fetch_graph":"https://pith.science/api/pith-number/WQEG76N3JINWH4UCPV3IWWYKFF/graph.json","fetch_events":"https://pith.science/api/pith-number/WQEG76N3JINWH4UCPV3IWWYKFF/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/WQEG76N3JINWH4UCPV3IWWYKFF/action/timestamp_anchor","attest_storage":"https://pith.science/pith/WQEG76N3JINWH4UCPV3IWWYKFF/action/storage_attestation","attest_author":"https://pith.science/pith/WQEG76N3JINWH4UCPV3IWWYKFF/action/author_attestation","sign_citation":"https://pith.science/pith/WQEG76N3JINWH4UCPV3IWWYKFF/action/citation_signature","submit_replication":"https://pith.science/pith/WQEG76N3JINWH4UCPV3IWWYKFF/action/replication_record"}},"created_at":"2026-07-07T00:15:52.252290+00:00","updated_at":"2026-07-07T00:15:52.252290+00:00"}