{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2020:PVRUE43MZF5LCJLDJOLEL6BFXT","short_pith_number":"pith:PVRUE43M","schema_version":"1.0","canonical_sha256":"7d6342736cc97ab125634b9645f825bcf99dfc7aedfc1e6ac2f1a7405165830c","source":{"kind":"arxiv","id":"2004.08249","version":3},"attestation_state":"computed","paper":{"title":"Understanding the Difficulty of Training Transformers","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL","stat.ML"],"primary_cat":"cs.LG","authors_text":"Jianfeng Gao, Jiawei Han, Liyuan Liu, Weizhu Chen, XiaoDong Liu","submitted_at":"2020-04-17T13:59:07Z","abstract_excerpt":"Transformers have proved effective in many NLP tasks. However, their training requires non-trivial efforts regarding designing cutting-edge optimizers and learning rate schedulers carefully (e.g., conventional SGD fails to train Transformers effectively). Our objective here is to understand $\\textit{what complicates Transformer training}$ from both empirical and theoretical perspectives. Our analysis reveals that unbalanced gradients are not the root cause of the instability of training. Instead, we identify an amplification effect that influences training substantially -- for each layer in a "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2004.08249","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2020-04-17T13:59:07Z","cross_cats_sorted":["cs.CL","stat.ML"],"title_canon_sha256":"2af21d92a6c95e5e2cd5dbb86c83a041b6a95c3dba1274b04912792efd1d4df8","abstract_canon_sha256":"c283bf9570da988330b1deda321fe4220bd2e3513c241079f322316b41ace8fc"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:55:45.986616Z","signature_b64":"dRLUICFjuL+QDEXV+MjKLFI8zGN1SAQ5eqt8mef+u8n8nwhrRrnfrXCzV+778IWLUlqGl4oKyNsuEp72dD8bAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"7d6342736cc97ab125634b9645f825bcf99dfc7aedfc1e6ac2f1a7405165830c","last_reissued_at":"2026-07-05T06:55:45.986200Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:55:45.986200Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Understanding the Difficulty of Training Transformers","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL","stat.ML"],"primary_cat":"cs.LG","authors_text":"Jianfeng Gao, Jiawei Han, Liyuan Liu, Weizhu Chen, XiaoDong Liu","submitted_at":"2020-04-17T13:59:07Z","abstract_excerpt":"Transformers have proved effective in many NLP tasks. However, their training requires non-trivial efforts regarding designing cutting-edge optimizers and learning rate schedulers carefully (e.g., conventional SGD fails to train Transformers effectively). Our objective here is to understand $\\textit{what complicates Transformer training}$ from both empirical and theoretical perspectives. Our analysis reveals that unbalanced gradients are not the root cause of the instability of training. Instead, we identify an amplification effect that influences training substantially -- for each layer in a "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2004.08249","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2004.08249/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2004.08249","created_at":"2026-07-05T06:55:45.986259+00:00"},{"alias_kind":"arxiv_version","alias_value":"2004.08249v3","created_at":"2026-07-05T06:55:45.986259+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2004.08249","created_at":"2026-07-05T06:55:45.986259+00:00"},{"alias_kind":"pith_short_12","alias_value":"PVRUE43MZF5L","created_at":"2026-07-05T06:55:45.986259+00:00"},{"alias_kind":"pith_short_16","alias_value":"PVRUE43MZF5LCJLD","created_at":"2026-07-05T06:55:45.986259+00:00"},{"alias_kind":"pith_short_8","alias_value":"PVRUE43M","created_at":"2026-07-05T06:55:45.986259+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.19797","citing_title":"Improving End-to-End Speech Recognition for Dysarthric Speech through In-Domain Data Augmentation","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2507.06261","citing_title":"Gemini 2.5: Pushing the Frontier with Advanced Reasoning, Multimodality, Long Context, and Next Generation Agentic Capabilities","ref_index":51,"is_internal_anchor":false},{"citing_arxiv_id":"2306.14048","citing_title":"H$_2$O: Heavy-Hitter Oracle for Efficient Generative Inference of Large Language Models","ref_index":83,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05794","citing_title":"Revealing Modular Gradient Noise Imbalance in LLMs: Calibrating Adam via Signal-to-Noise Ratio","ref_index":19,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/PVRUE43MZF5LCJLDJOLEL6BFXT","json":"https://pith.science/pith/PVRUE43MZF5LCJLDJOLEL6BFXT.json","graph_json":"https://pith.science/api/pith-number/PVRUE43MZF5LCJLDJOLEL6BFXT/graph.json","events_json":"https://pith.science/api/pith-number/PVRUE43MZF5LCJLDJOLEL6BFXT/events.json","paper":"https://pith.science/paper/PVRUE43M"},"agent_actions":{"view_html":"https://pith.science/pith/PVRUE43MZF5LCJLDJOLEL6BFXT","download_json":"https://pith.science/pith/PVRUE43MZF5LCJLDJOLEL6BFXT.json","view_paper":"https://pith.science/paper/PVRUE43M","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2004.08249&json=true","fetch_graph":"https://pith.science/api/pith-number/PVRUE43MZF5LCJLDJOLEL6BFXT/graph.json","fetch_events":"https://pith.science/api/pith-number/PVRUE43MZF5LCJLDJOLEL6BFXT/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/PVRUE43MZF5LCJLDJOLEL6BFXT/action/timestamp_anchor","attest_storage":"https://pith.science/pith/PVRUE43MZF5LCJLDJOLEL6BFXT/action/storage_attestation","attest_author":"https://pith.science/pith/PVRUE43MZF5LCJLDJOLEL6BFXT/action/author_attestation","sign_citation":"https://pith.science/pith/PVRUE43MZF5LCJLDJOLEL6BFXT/action/citation_signature","submit_replication":"https://pith.science/pith/PVRUE43MZF5LCJLDJOLEL6BFXT/action/replication_record"}},"created_at":"2026-07-05T06:55:45.986259+00:00","updated_at":"2026-07-05T06:55:45.986259+00:00"}