{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:BKVPYDMBARKNTNKKREA6FZCWV5","short_pith_number":"pith:BKVPYDMB","schema_version":"1.0","canonical_sha256":"0aaafc0d810454d9b54a8901e2e456af498d89440f2263dacf5d03df38d7bd56","source":{"kind":"arxiv","id":"2307.15936","version":2},"attestation_state":"computed","paper":{"title":"A Theory for Emergence of Complex Skills in Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL","stat.ML"],"primary_cat":"cs.LG","authors_text":"Anirudh Goyal, Sanjeev Arora","submitted_at":"2023-07-29T09:22:54Z","abstract_excerpt":"A major driver of AI products today is the fact that new skills emerge in language models when their parameter set and training corpora are scaled up. This phenomenon is poorly understood, and a mechanistic explanation via mathematical analysis of gradient-based training seems difficult. The current paper takes a different approach, analysing emergence using the famous (and empirical) Scaling Laws of LLMs and a simple statistical framework. Contributions include: (a) A statistical framework that relates cross-entropy loss of LLMs to competence on the basic skills that underlie language tasks. "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2307.15936","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2023-07-29T09:22:54Z","cross_cats_sorted":["cs.AI","cs.CL","stat.ML"],"title_canon_sha256":"e363eb8c2fd8d0c7abea2cd4eaf2ab1a15b202156ad746f8ca935f4751123585","abstract_canon_sha256":"fab594f2c5557405b2d9654fecab1c8afd0c3e94f3c3f39990a2c1931161fea4"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:09:00.353781Z","signature_b64":"XppgT6N/Lk3zq2hl6m6Ar0+t4J/cC/evty2G7pyF+Ody/2FycMTtAUje3oCtae2t1/B8U2baQqN9AallNsL+Bg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"0aaafc0d810454d9b54a8901e2e456af498d89440f2263dacf5d03df38d7bd56","last_reissued_at":"2026-07-05T07:09:00.353326Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:09:00.353326Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"A Theory for Emergence of Complex Skills in Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL","stat.ML"],"primary_cat":"cs.LG","authors_text":"Anirudh Goyal, Sanjeev Arora","submitted_at":"2023-07-29T09:22:54Z","abstract_excerpt":"A major driver of AI products today is the fact that new skills emerge in language models when their parameter set and training corpora are scaled up. This phenomenon is poorly understood, and a mechanistic explanation via mathematical analysis of gradient-based training seems difficult. The current paper takes a different approach, analysing emergence using the famous (and empirical) Scaling Laws of LLMs and a simple statistical framework. Contributions include: (a) A statistical framework that relates cross-entropy loss of LLMs to competence on the basic skills that underlie language tasks. "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2307.15936","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2307.15936/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2307.15936","created_at":"2026-07-05T07:09:00.353390+00:00"},{"alias_kind":"arxiv_version","alias_value":"2307.15936v2","created_at":"2026-07-05T07:09:00.353390+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2307.15936","created_at":"2026-07-05T07:09:00.353390+00:00"},{"alias_kind":"pith_short_12","alias_value":"BKVPYDMBARKN","created_at":"2026-07-05T07:09:00.353390+00:00"},{"alias_kind":"pith_short_16","alias_value":"BKVPYDMBARKNTNKK","created_at":"2026-07-05T07:09:00.353390+00:00"},{"alias_kind":"pith_short_8","alias_value":"BKVPYDMB","created_at":"2026-07-05T07:09:00.353390+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":15,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.25008","citing_title":"Neural Scaling Universality: If Exponents Are Fixed, Time to Understand Coefficients","ref_index":40,"is_internal_anchor":false},{"citing_arxiv_id":"2607.01647","citing_title":"AgenticDataBench: A Comprehensive Benchmark for Data Agents","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2606.08167","citing_title":"Explaining Data Mixing Scaling Laws","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2606.07568","citing_title":"A Systematic Study of Behavioral Cloning for Scientific Data Annotation","ref_index":217,"is_internal_anchor":false},{"citing_arxiv_id":"2605.29548","citing_title":"Why Larger Models Learn More: Effects of Capacity, Interference, and Rare-Task Retention","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17767","citing_title":"Feature Learning in Linear-Width Two-Layer Networks: Two vs. One Step of Gradient Descent","ref_index":221,"is_internal_anchor":false},{"citing_arxiv_id":"2501.02378","citing_title":"A ghost mechanism: An analytical model of abrupt learning in recurrent networks","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2605.04970","citing_title":"Skill Neologisms: Towards Skill-based Continual Learning","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17767","citing_title":"Feature Learning in Linear-Width Two-Layer Networks: Two vs. One Step of Gradient Descent","ref_index":221,"is_internal_anchor":false},{"citing_arxiv_id":"2605.13612","citing_title":"Deep Learning as Neural Low-Degree Filtering: A Spectral Theory of Hierarchical Feature Learning","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2604.24294","citing_title":"AI-Native Autonomous Infrastructure (ANAI): A Formal Framework for the Next General-Purpose Technology","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2604.22951","citing_title":"The Power of Power Law: Asymmetry Enables Compositional Reasoning","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2605.04970","citing_title":"Skill Neologisms: Towards Skill-based Continual Learning","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01034","citing_title":"A Theoretical Game of Attacks via Compositional Skills","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2604.17614","citing_title":"Characterizing Model-Native Skills","ref_index":2,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/BKVPYDMBARKNTNKKREA6FZCWV5","json":"https://pith.science/pith/BKVPYDMBARKNTNKKREA6FZCWV5.json","graph_json":"https://pith.science/api/pith-number/BKVPYDMBARKNTNKKREA6FZCWV5/graph.json","events_json":"https://pith.science/api/pith-number/BKVPYDMBARKNTNKKREA6FZCWV5/events.json","paper":"https://pith.science/paper/BKVPYDMB"},"agent_actions":{"view_html":"https://pith.science/pith/BKVPYDMBARKNTNKKREA6FZCWV5","download_json":"https://pith.science/pith/BKVPYDMBARKNTNKKREA6FZCWV5.json","view_paper":"https://pith.science/paper/BKVPYDMB","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2307.15936&json=true","fetch_graph":"https://pith.science/api/pith-number/BKVPYDMBARKNTNKKREA6FZCWV5/graph.json","fetch_events":"https://pith.science/api/pith-number/BKVPYDMBARKNTNKKREA6FZCWV5/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/BKVPYDMBARKNTNKKREA6FZCWV5/action/timestamp_anchor","attest_storage":"https://pith.science/pith/BKVPYDMBARKNTNKKREA6FZCWV5/action/storage_attestation","attest_author":"https://pith.science/pith/BKVPYDMBARKNTNKKREA6FZCWV5/action/author_attestation","sign_citation":"https://pith.science/pith/BKVPYDMBARKNTNKKREA6FZCWV5/action/citation_signature","submit_replication":"https://pith.science/pith/BKVPYDMBARKNTNKKREA6FZCWV5/action/replication_record"}},"created_at":"2026-07-05T07:09:00.353390+00:00","updated_at":"2026-07-05T07:09:00.353390+00:00"}