{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:ABTJTQXZFN2JTHUBLYI6GA3ULP","short_pith_number":"pith:ABTJTQXZ","schema_version":"1.0","canonical_sha256":"006699c2f92b74999e815e11e303745bf68abaa620a7bb913c91957399f420cb","source":{"kind":"arxiv","id":"2212.09803","version":3},"attestation_state":"computed","paper":{"title":"Training Trajectories of Language Models Across Scales","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Chunting Zhou, Danqi Chen, Luke Zettlemoyer, Mengzhou Xia, Mikel Artetxe, Ramakanth Pasunuru, Ves Stoyanov, Xi Victoria Lin","submitted_at":"2022-12-19T19:16:29Z","abstract_excerpt":"Scaling up language models has led to unprecedented performance gains, but little is understood about how the training dynamics change as models get larger. How do language models of different sizes learn during pre-training? Why do larger language models demonstrate more desirable behaviors? In this paper, we analyze the intermediate training checkpoints of differently sized OPT models (Zhang et al.,2022)--from 125M to 175B parameters--on next-token prediction, sequence-level generation, and downstream tasks. We find that 1) at a given perplexity and independent of model sizes, a similar subs"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2212.09803","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2022-12-19T19:16:29Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"624666ea8a2bb18baf38ac5ec844c51702b3a789eda9eb412918fcb3f8e24962","abstract_canon_sha256":"6378b83e99431018c068361c9e2db9b4553188c30e12bb74358d50d5240f6540"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:14:56.995781Z","signature_b64":"jc1wXuuyhK0Yq1ep8m6MLUOoLDedCQn/rdMT+6h1MEaI1alKcWVbWf9m3MPTQsnmUSRVESWB/20YmH378U4ZDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"006699c2f92b74999e815e11e303745bf68abaa620a7bb913c91957399f420cb","last_reissued_at":"2026-07-05T06:14:56.995371Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:14:56.995371Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Training Trajectories of Language Models Across Scales","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Chunting Zhou, Danqi Chen, Luke Zettlemoyer, Mengzhou Xia, Mikel Artetxe, Ramakanth Pasunuru, Ves Stoyanov, Xi Victoria Lin","submitted_at":"2022-12-19T19:16:29Z","abstract_excerpt":"Scaling up language models has led to unprecedented performance gains, but little is understood about how the training dynamics change as models get larger. How do language models of different sizes learn during pre-training? Why do larger language models demonstrate more desirable behaviors? In this paper, we analyze the intermediate training checkpoints of differently sized OPT models (Zhang et al.,2022)--from 125M to 175B parameters--on next-token prediction, sequence-level generation, and downstream tasks. We find that 1) at a given perplexity and independent of model sizes, a similar subs"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2212.09803","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2212.09803/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2212.09803","created_at":"2026-07-05T06:14:56.995429+00:00"},{"alias_kind":"arxiv_version","alias_value":"2212.09803v3","created_at":"2026-07-05T06:14:56.995429+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2212.09803","created_at":"2026-07-05T06:14:56.995429+00:00"},{"alias_kind":"pith_short_12","alias_value":"ABTJTQXZFN2J","created_at":"2026-07-05T06:14:56.995429+00:00"},{"alias_kind":"pith_short_16","alias_value":"ABTJTQXZFN2JTHUB","created_at":"2026-07-05T06:14:56.995429+00:00"},{"alias_kind":"pith_short_8","alias_value":"ABTJTQXZ","created_at":"2026-07-05T06:14:56.995429+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.17767","citing_title":"Feature Learning in Linear-Width Two-Layer Networks: Two vs. One Step of Gradient Descent","ref_index":226,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17767","citing_title":"Feature Learning in Linear-Width Two-Layer Networks: Two vs. One Step of Gradient Descent","ref_index":226,"is_internal_anchor":false},{"citing_arxiv_id":"2502.10248","citing_title":"Step-Video-T2V Technical Report: The Practice, Challenges, and Future of Video Foundation Model","ref_index":219,"is_internal_anchor":false},{"citing_arxiv_id":"2305.16264","citing_title":"Scaling Data-Constrained Language Models","ref_index":131,"is_internal_anchor":false},{"citing_arxiv_id":"2304.01373","citing_title":"Pythia: A Suite for Analyzing Large Language Models Across Training and Scaling","ref_index":91,"is_internal_anchor":false},{"citing_arxiv_id":"2304.01373","citing_title":"Pythia: A Suite for Analyzing Large Language Models Across Training and Scaling","ref_index":180,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ABTJTQXZFN2JTHUBLYI6GA3ULP","json":"https://pith.science/pith/ABTJTQXZFN2JTHUBLYI6GA3ULP.json","graph_json":"https://pith.science/api/pith-number/ABTJTQXZFN2JTHUBLYI6GA3ULP/graph.json","events_json":"https://pith.science/api/pith-number/ABTJTQXZFN2JTHUBLYI6GA3ULP/events.json","paper":"https://pith.science/paper/ABTJTQXZ"},"agent_actions":{"view_html":"https://pith.science/pith/ABTJTQXZFN2JTHUBLYI6GA3ULP","download_json":"https://pith.science/pith/ABTJTQXZFN2JTHUBLYI6GA3ULP.json","view_paper":"https://pith.science/paper/ABTJTQXZ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2212.09803&json=true","fetch_graph":"https://pith.science/api/pith-number/ABTJTQXZFN2JTHUBLYI6GA3ULP/graph.json","fetch_events":"https://pith.science/api/pith-number/ABTJTQXZFN2JTHUBLYI6GA3ULP/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ABTJTQXZFN2JTHUBLYI6GA3ULP/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ABTJTQXZFN2JTHUBLYI6GA3ULP/action/storage_attestation","attest_author":"https://pith.science/pith/ABTJTQXZFN2JTHUBLYI6GA3ULP/action/author_attestation","sign_citation":"https://pith.science/pith/ABTJTQXZFN2JTHUBLYI6GA3ULP/action/citation_signature","submit_replication":"https://pith.science/pith/ABTJTQXZFN2JTHUBLYI6GA3ULP/action/replication_record"}},"created_at":"2026-07-05T06:14:56.995429+00:00","updated_at":"2026-07-05T06:14:56.995429+00:00"}