{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:XHXDT54HGMTMD3AWS7F2BF6UOG","short_pith_number":"pith:XHXDT54H","schema_version":"1.0","canonical_sha256":"b9ee39f7873326c1ec1697cba097d47195acac6725685084dd8d08711704604f","source":{"kind":"arxiv","id":"2308.15419","version":2},"attestation_state":"computed","paper":{"title":"Characterizing Learning Curves During Language Model Pre-Training: Learning, Forgetting, and Stability","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Benjamin K. Bergen, Tyler A. Chang, Zhuowen Tu","submitted_at":"2023-08-29T16:24:09Z","abstract_excerpt":"How do language models learn to make predictions during pre-training? To study this, we extract learning curves from five autoregressive English language model pre-training runs, for 1M unseen tokens in context. We observe that the language models generate short repetitive phrases before learning to generate longer and more coherent text. We also find that individual tokens often exhibit sudden increases or decreases in loss that are surprisingly consistent across pre-training runs. To better understand these fluctuations, we quantify the final surprisal, within-run variability, age of acquisi"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2308.15419","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2023-08-29T16:24:09Z","cross_cats_sorted":[],"title_canon_sha256":"a144f1184dfd5d63fc4b5e57c85838747cd1c592e17357402cf2362a4dab7b6b","abstract_canon_sha256":"d5645bbfde70c6dd810e3647360470353c1bb0d05704af1a51d43cdbbe559377"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:50:24.382515Z","signature_b64":"fmnBD+Mp8E55fSbOrRd2dvebvSMRUMm+n9h3nDxMqoVAB8/yrr6kP/5+YQnbUbTzT7g/JtR4W5B6HdHVEaNQAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b9ee39f7873326c1ec1697cba097d47195acac6725685084dd8d08711704604f","last_reissued_at":"2026-07-05T08:50:24.382084Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:50:24.382084Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Characterizing Learning Curves During Language Model Pre-Training: Learning, Forgetting, and Stability","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Benjamin K. Bergen, Tyler A. Chang, Zhuowen Tu","submitted_at":"2023-08-29T16:24:09Z","abstract_excerpt":"How do language models learn to make predictions during pre-training? To study this, we extract learning curves from five autoregressive English language model pre-training runs, for 1M unseen tokens in context. We observe that the language models generate short repetitive phrases before learning to generate longer and more coherent text. We also find that individual tokens often exhibit sudden increases or decreases in loss that are surprisingly consistent across pre-training runs. To better understand these fluctuations, we quantify the final surprisal, within-run variability, age of acquisi"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2308.15419","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2308.15419/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2308.15419","created_at":"2026-07-05T08:50:24.382139+00:00"},{"alias_kind":"arxiv_version","alias_value":"2308.15419v2","created_at":"2026-07-05T08:50:24.382139+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2308.15419","created_at":"2026-07-05T08:50:24.382139+00:00"},{"alias_kind":"pith_short_12","alias_value":"XHXDT54HGMTM","created_at":"2026-07-05T08:50:24.382139+00:00"},{"alias_kind":"pith_short_16","alias_value":"XHXDT54HGMTMD3AW","created_at":"2026-07-05T08:50:24.382139+00:00"},{"alias_kind":"pith_short_8","alias_value":"XHXDT54H","created_at":"2026-07-05T08:50:24.382139+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2506.03434","citing_title":"Time Course MechInterp: Analyzing the Evolution of Components and Knowledge in Large Language Models","ref_index":2023,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/XHXDT54HGMTMD3AWS7F2BF6UOG","json":"https://pith.science/pith/XHXDT54HGMTMD3AWS7F2BF6UOG.json","graph_json":"https://pith.science/api/pith-number/XHXDT54HGMTMD3AWS7F2BF6UOG/graph.json","events_json":"https://pith.science/api/pith-number/XHXDT54HGMTMD3AWS7F2BF6UOG/events.json","paper":"https://pith.science/paper/XHXDT54H"},"agent_actions":{"view_html":"https://pith.science/pith/XHXDT54HGMTMD3AWS7F2BF6UOG","download_json":"https://pith.science/pith/XHXDT54HGMTMD3AWS7F2BF6UOG.json","view_paper":"https://pith.science/paper/XHXDT54H","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2308.15419&json=true","fetch_graph":"https://pith.science/api/pith-number/XHXDT54HGMTMD3AWS7F2BF6UOG/graph.json","fetch_events":"https://pith.science/api/pith-number/XHXDT54HGMTMD3AWS7F2BF6UOG/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/XHXDT54HGMTMD3AWS7F2BF6UOG/action/timestamp_anchor","attest_storage":"https://pith.science/pith/XHXDT54HGMTMD3AWS7F2BF6UOG/action/storage_attestation","attest_author":"https://pith.science/pith/XHXDT54HGMTMD3AWS7F2BF6UOG/action/author_attestation","sign_citation":"https://pith.science/pith/XHXDT54HGMTMD3AWS7F2BF6UOG/action/citation_signature","submit_replication":"https://pith.science/pith/XHXDT54HGMTMD3AWS7F2BF6UOG/action/replication_record"}},"created_at":"2026-07-05T08:50:24.382139+00:00","updated_at":"2026-07-05T08:50:24.382139+00:00"}