{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2020:XO623QCLIJ4IAJZGR2EGCVU5CW","short_pith_number":"pith:XO623QCL","schema_version":"1.0","canonical_sha256":"bbbdadc04b42788027268e8861569d15ba978ae712466afef159e693bbe575a2","source":{"kind":"arxiv","id":"2010.09697","version":5},"attestation_state":"computed","paper":{"title":"Effects of Parameter Norm Growth During Transformer Training: Inductive Bias from Gradient Descent","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.LG","authors_text":"Noah Smith, Roy Schwartz, Vivek Ramanujan, William Merrill, Yoav Goldberg","submitted_at":"2020-10-19T17:40:38Z","abstract_excerpt":"The capacity of neural networks like the widely adopted transformer is known to be very high. Evidence is emerging that they learn successfully due to inductive bias in the training routine, typically a variant of gradient descent (GD). To better understand this bias, we study the tendency for transformer parameters to grow in magnitude ($\\ell_2$ norm) during training, and its implications for the emergent representations within self attention layers. Empirically, we document norm growth in the training of transformer language models, including T5 during its pretraining. As the parameters grow"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2010.09697","kind":"arxiv","version":5},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2020-10-19T17:40:38Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"42420202fe1aee0fdc109f45599d08ec4d72d46d584bf8a99b338dda8276de53","abstract_canon_sha256":"7952a381eb584be59494641d30eef823d6d0c91bf49484db9e3108a340afd8b5"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:49:11.294422Z","signature_b64":"KNvnyRCm72AJamPc7uS9DYSVG4fvyC8D3nb56ADTU09iAizkbeRPdXXONb0wQkT9Ftr8XctpoNs1YqjnMgw8Cw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"bbbdadc04b42788027268e8861569d15ba978ae712466afef159e693bbe575a2","last_reissued_at":"2026-07-05T05:49:11.293772Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:49:11.293772Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Effects of Parameter Norm Growth During Transformer Training: Inductive Bias from Gradient Descent","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.LG","authors_text":"Noah Smith, Roy Schwartz, Vivek Ramanujan, William Merrill, Yoav Goldberg","submitted_at":"2020-10-19T17:40:38Z","abstract_excerpt":"The capacity of neural networks like the widely adopted transformer is known to be very high. Evidence is emerging that they learn successfully due to inductive bias in the training routine, typically a variant of gradient descent (GD). To better understand this bias, we study the tendency for transformer parameters to grow in magnitude ($\\ell_2$ norm) during training, and its implications for the emergent representations within self attention layers. Empirically, we document norm growth in the training of transformer language models, including T5 during its pretraining. As the parameters grow"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2010.09697","kind":"arxiv","version":5},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2010.09697/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2010.09697","created_at":"2026-07-05T05:49:11.293835+00:00"},{"alias_kind":"arxiv_version","alias_value":"2010.09697v5","created_at":"2026-07-05T05:49:11.293835+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2010.09697","created_at":"2026-07-05T05:49:11.293835+00:00"},{"alias_kind":"pith_short_12","alias_value":"XO623QCLIJ4I","created_at":"2026-07-05T05:49:11.293835+00:00"},{"alias_kind":"pith_short_16","alias_value":"XO623QCLIJ4IAJZG","created_at":"2026-07-05T05:49:11.293835+00:00"},{"alias_kind":"pith_short_8","alias_value":"XO623QCL","created_at":"2026-07-05T05:49:11.293835+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2501.04697","citing_title":"Grokking at the Edge of Numerical Stability","ref_index":9,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/XO623QCLIJ4IAJZGR2EGCVU5CW","json":"https://pith.science/pith/XO623QCLIJ4IAJZGR2EGCVU5CW.json","graph_json":"https://pith.science/api/pith-number/XO623QCLIJ4IAJZGR2EGCVU5CW/graph.json","events_json":"https://pith.science/api/pith-number/XO623QCLIJ4IAJZGR2EGCVU5CW/events.json","paper":"https://pith.science/paper/XO623QCL"},"agent_actions":{"view_html":"https://pith.science/pith/XO623QCLIJ4IAJZGR2EGCVU5CW","download_json":"https://pith.science/pith/XO623QCLIJ4IAJZGR2EGCVU5CW.json","view_paper":"https://pith.science/paper/XO623QCL","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2010.09697&json=true","fetch_graph":"https://pith.science/api/pith-number/XO623QCLIJ4IAJZGR2EGCVU5CW/graph.json","fetch_events":"https://pith.science/api/pith-number/XO623QCLIJ4IAJZGR2EGCVU5CW/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/XO623QCLIJ4IAJZGR2EGCVU5CW/action/timestamp_anchor","attest_storage":"https://pith.science/pith/XO623QCLIJ4IAJZGR2EGCVU5CW/action/storage_attestation","attest_author":"https://pith.science/pith/XO623QCLIJ4IAJZGR2EGCVU5CW/action/author_attestation","sign_citation":"https://pith.science/pith/XO623QCLIJ4IAJZGR2EGCVU5CW/action/citation_signature","submit_replication":"https://pith.science/pith/XO623QCLIJ4IAJZGR2EGCVU5CW/action/replication_record"}},"created_at":"2026-07-05T05:49:11.293835+00:00","updated_at":"2026-07-05T05:49:11.293835+00:00"}