{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:SFCKLZ65NRC3FW4P3NIP3PC3K3","short_pith_number":"pith:SFCKLZ65","schema_version":"1.0","canonical_sha256":"9144a5e7dd6c45b2db8fdb50fdbc5b56c9e92e577d06ad67bb038c307ea0957d","source":{"kind":"arxiv","id":"2312.16903","version":4},"attestation_state":"computed","paper":{"title":"Spike No More: Stabilizing the Pre-training of Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Jun Suzuki, Sho Takase, Shun Kiyono, Sosuke Kobayashi","submitted_at":"2023-12-28T08:53:27Z","abstract_excerpt":"Loss spikes often occur during pre-training of large language models. The spikes degrade the performance of large language models and sometimes ruin the pre-training. Since the pre-training needs a vast computational budget, we should avoid such spikes. Based on the assumption that the loss spike is caused by the sudden growth of the gradient norm, we explore factors to keep the gradient norm small through an analysis of the spectral norms of the Jacobian matrices for the sub-layers. Our findings suggest that stabilizing the pre-training process requires two conditions: small sub-layers and la"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2312.16903","kind":"arxiv","version":4},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2023-12-28T08:53:27Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"100a2c61fc9db602e6e65cfbe0324ea38bffa4abeeb0d129e49264d78f1a6d33","abstract_canon_sha256":"a11e59b47a59532ef6093f1f35dc5517f0fd5b850939a921e85862470ba26502"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:43:05.608011Z","signature_b64":"upyCeXNUFJxAYHGD72oHEt8eth/FG0lsO5wuBwuktPrNoIcpH1lyNK9oAidqW68YuM/YlMZUypDDyN48ATr5CA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"9144a5e7dd6c45b2db8fdb50fdbc5b56c9e92e577d06ad67bb038c307ea0957d","last_reissued_at":"2026-07-05T11:43:05.607541Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:43:05.607541Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Spike No More: Stabilizing the Pre-training of Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Jun Suzuki, Sho Takase, Shun Kiyono, Sosuke Kobayashi","submitted_at":"2023-12-28T08:53:27Z","abstract_excerpt":"Loss spikes often occur during pre-training of large language models. The spikes degrade the performance of large language models and sometimes ruin the pre-training. Since the pre-training needs a vast computational budget, we should avoid such spikes. Based on the assumption that the loss spike is caused by the sudden growth of the gradient norm, we explore factors to keep the gradient norm small through an analysis of the spectral norms of the Jacobian matrices for the sub-layers. Our findings suggest that stabilizing the pre-training process requires two conditions: small sub-layers and la"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2312.16903","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2312.16903/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2312.16903","created_at":"2026-07-05T11:43:05.607601+00:00"},{"alias_kind":"arxiv_version","alias_value":"2312.16903v4","created_at":"2026-07-05T11:43:05.607601+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2312.16903","created_at":"2026-07-05T11:43:05.607601+00:00"},{"alias_kind":"pith_short_12","alias_value":"SFCKLZ65NRC3","created_at":"2026-07-05T11:43:05.607601+00:00"},{"alias_kind":"pith_short_16","alias_value":"SFCKLZ65NRC3FW4P","created_at":"2026-07-05T11:43:05.607601+00:00"},{"alias_kind":"pith_short_8","alias_value":"SFCKLZ65","created_at":"2026-07-05T11:43:05.607601+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2607.01775","citing_title":"Set Diffusion: Interpolating Token Orderings Between Autoregression and Diffusion for Fast and Flexible Decoding","ref_index":99,"is_internal_anchor":false},{"citing_arxiv_id":"2606.00539","citing_title":"GNMR: Runtime Stability Control for Low-Precision Large Language Model Training","ref_index":49,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17787","citing_title":"Revisiting the Adam-SGD Gap in LLM Pre-Training: The Role of Large Effective Learning Rates","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2510.18900","citing_title":"Foundation Models for Discovery and Exploration in Chemical Space","ref_index":146,"is_internal_anchor":false},{"citing_arxiv_id":"2502.05171","citing_title":"Scaling up Test-Time Compute with Latent Reasoning: A Recurrent Depth Approach","ref_index":154,"is_internal_anchor":false},{"citing_arxiv_id":"2604.12946","citing_title":"Parcae: Scaling Laws For Stable Looped Language Models","ref_index":76,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/SFCKLZ65NRC3FW4P3NIP3PC3K3","json":"https://pith.science/pith/SFCKLZ65NRC3FW4P3NIP3PC3K3.json","graph_json":"https://pith.science/api/pith-number/SFCKLZ65NRC3FW4P3NIP3PC3K3/graph.json","events_json":"https://pith.science/api/pith-number/SFCKLZ65NRC3FW4P3NIP3PC3K3/events.json","paper":"https://pith.science/paper/SFCKLZ65"},"agent_actions":{"view_html":"https://pith.science/pith/SFCKLZ65NRC3FW4P3NIP3PC3K3","download_json":"https://pith.science/pith/SFCKLZ65NRC3FW4P3NIP3PC3K3.json","view_paper":"https://pith.science/paper/SFCKLZ65","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2312.16903&json=true","fetch_graph":"https://pith.science/api/pith-number/SFCKLZ65NRC3FW4P3NIP3PC3K3/graph.json","fetch_events":"https://pith.science/api/pith-number/SFCKLZ65NRC3FW4P3NIP3PC3K3/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/SFCKLZ65NRC3FW4P3NIP3PC3K3/action/timestamp_anchor","attest_storage":"https://pith.science/pith/SFCKLZ65NRC3FW4P3NIP3PC3K3/action/storage_attestation","attest_author":"https://pith.science/pith/SFCKLZ65NRC3FW4P3NIP3PC3K3/action/author_attestation","sign_citation":"https://pith.science/pith/SFCKLZ65NRC3FW4P3NIP3PC3K3/action/citation_signature","submit_replication":"https://pith.science/pith/SFCKLZ65NRC3FW4P3NIP3PC3K3/action/replication_record"}},"created_at":"2026-07-05T11:43:05.607601+00:00","updated_at":"2026-07-05T11:43:05.607601+00:00"}