{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:E5BC33AOZ6L7DA3D5QEHQBULXS","short_pith_number":"pith:E5BC33AO","schema_version":"1.0","canonical_sha256":"27422dec0ecf97f18363ec0878068bbcaf8daa5c147f18daeed81db724957f44","source":{"kind":"arxiv","id":"2410.16682","version":1},"attestation_state":"computed","paper":{"title":"Methods of improving LLM training stability","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Ben Lanir, Jinze Xue, Mike Chrzanowski, Oleg Rybakov, Peter Dykas","submitted_at":"2024-10-22T04:27:03Z","abstract_excerpt":"Training stability of large language models(LLMs) is an important research topic. Reproducing training instabilities can be costly, so we use a small language model with 830M parameters and experiment with higher learning rates to force models to diverge. One of the sources of training instability is the growth of logits in attention layers. We extend the focus of the previous work and look not only at the magnitude of the logits but at all outputs of linear layers in the Transformer block. We observe that with a high learning rate the L2 norm of all linear layer outputs can grow with each tra"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.16682","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-10-22T04:27:03Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"2aedd500fd8f6c90677746a26a4d139cd0cbf23548a13ae4191eb5ce4022276e","abstract_canon_sha256":"478f129a95a473ffff0dada3fecfeef4564175a9952eea039e168db9bc4e5b30"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:24:05.119132Z","signature_b64":"tLkfwu1EtBaVfZzNYwAvOID+bDqaZYo+yykJGzu2R3PGV3vuaB5DDZrL+bTrxSt/7OVrRzo+yiyKl9atF3ALBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"27422dec0ecf97f18363ec0878068bbcaf8daa5c147f18daeed81db724957f44","last_reissued_at":"2026-07-05T09:24:05.118607Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:24:05.118607Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Methods of improving LLM training stability","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Ben Lanir, Jinze Xue, Mike Chrzanowski, Oleg Rybakov, Peter Dykas","submitted_at":"2024-10-22T04:27:03Z","abstract_excerpt":"Training stability of large language models(LLMs) is an important research topic. Reproducing training instabilities can be costly, so we use a small language model with 830M parameters and experiment with higher learning rates to force models to diverge. One of the sources of training instability is the growth of logits in attention layers. We extend the focus of the previous work and look not only at the magnitude of the logits but at all outputs of linear layers in the Transformer block. We observe that with a high learning rate the L2 norm of all linear layer outputs can grow with each tra"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.16682","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.16682/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.16682","created_at":"2026-07-05T09:24:05.118679+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.16682v1","created_at":"2026-07-05T09:24:05.118679+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.16682","created_at":"2026-07-05T09:24:05.118679+00:00"},{"alias_kind":"pith_short_12","alias_value":"E5BC33AOZ6L7","created_at":"2026-07-05T09:24:05.118679+00:00"},{"alias_kind":"pith_short_16","alias_value":"E5BC33AOZ6L7DA3D","created_at":"2026-07-05T09:24:05.118679+00:00"},{"alias_kind":"pith_short_8","alias_value":"E5BC33AO","created_at":"2026-07-05T09:24:05.118679+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.08105","citing_title":"A Unifying View of Attention Sinks: Two Algorithms, Two Solutions","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2510.04212","citing_title":"Why Low-Precision Transformer Training Fails: An Analysis on Flash Attention","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12492","citing_title":"Pion: A Spectrum-Preserving Optimizer via Orthogonal Equivalence Transformation","ref_index":66,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/E5BC33AOZ6L7DA3D5QEHQBULXS","json":"https://pith.science/pith/E5BC33AOZ6L7DA3D5QEHQBULXS.json","graph_json":"https://pith.science/api/pith-number/E5BC33AOZ6L7DA3D5QEHQBULXS/graph.json","events_json":"https://pith.science/api/pith-number/E5BC33AOZ6L7DA3D5QEHQBULXS/events.json","paper":"https://pith.science/paper/E5BC33AO"},"agent_actions":{"view_html":"https://pith.science/pith/E5BC33AOZ6L7DA3D5QEHQBULXS","download_json":"https://pith.science/pith/E5BC33AOZ6L7DA3D5QEHQBULXS.json","view_paper":"https://pith.science/paper/E5BC33AO","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.16682&json=true","fetch_graph":"https://pith.science/api/pith-number/E5BC33AOZ6L7DA3D5QEHQBULXS/graph.json","fetch_events":"https://pith.science/api/pith-number/E5BC33AOZ6L7DA3D5QEHQBULXS/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/E5BC33AOZ6L7DA3D5QEHQBULXS/action/timestamp_anchor","attest_storage":"https://pith.science/pith/E5BC33AOZ6L7DA3D5QEHQBULXS/action/storage_attestation","attest_author":"https://pith.science/pith/E5BC33AOZ6L7DA3D5QEHQBULXS/action/author_attestation","sign_citation":"https://pith.science/pith/E5BC33AOZ6L7DA3D5QEHQBULXS/action/citation_signature","submit_replication":"https://pith.science/pith/E5BC33AOZ6L7DA3D5QEHQBULXS/action/replication_record"}},"created_at":"2026-07-05T09:24:05.118679+00:00","updated_at":"2026-07-05T09:24:05.118679+00:00"}