{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2019:EYSLPSUYN3UBYVTQ6M7QQTMY5E","short_pith_number":"pith:EYSLPSUY","schema_version":"1.0","canonical_sha256":"2624b7ca986ee81c5670f33f084d98e93d806859fcf4646538a5e3617be2081d","source":{"kind":"arxiv","id":"1908.11365","version":1},"attestation_state":"computed","paper":{"title":"Improving Deep Transformer with Depth-Scaled Initialization and Merged Attention","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Biao Zhang, Ivan Titov, Rico Sennrich","submitted_at":"2019-08-29T17:50:55Z","abstract_excerpt":"The general trend in NLP is towards increasing model capacity and performance via deeper neural networks. However, simply stacking more layers of the popular Transformer architecture for machine translation results in poor convergence and high computational overhead. Our empirical analysis suggests that convergence is poor due to gradient vanishing caused by the interaction between residual connections and layer normalization. We propose depth-scaled initialization (DS-Init), which decreases parameter variance at the initialization stage, and reduces output variance of residual connections so "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"1908.11365","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2019-08-29T17:50:55Z","cross_cats_sorted":[],"title_canon_sha256":"da957cef716c564ce4e96e9c917e2dadb2e3c710ef6b05e240b5f64eb17f61a1","abstract_canon_sha256":"190ea37cc1869e7695b9a07f299f06967b08588fb1e2d455bfee9f22ea307d3d"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T00:00:32.914021Z","signature_b64":"UDZy6KjrXSNMHlhplwAuCzuzEbr2pWusP/BG+LRVQupi2fVcGUo9jgIjeIlMWw2poW0dFQ2jhkrxBMX/JwDuBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"2624b7ca986ee81c5670f33f084d98e93d806859fcf4646538a5e3617be2081d","last_reissued_at":"2026-07-05T00:00:32.913561Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T00:00:32.913561Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Improving Deep Transformer with Depth-Scaled Initialization and Merged Attention","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Biao Zhang, Ivan Titov, Rico Sennrich","submitted_at":"2019-08-29T17:50:55Z","abstract_excerpt":"The general trend in NLP is towards increasing model capacity and performance via deeper neural networks. However, simply stacking more layers of the popular Transformer architecture for machine translation results in poor convergence and high computational overhead. Our empirical analysis suggests that convergence is poor due to gradient vanishing caused by the interaction between residual connections and layer normalization. We propose depth-scaled initialization (DS-Init), which decreases parameter variance at the initialization stage, and reduces output variance of residual connections so "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1908.11365","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/1908.11365/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"1908.11365","created_at":"2026-07-05T00:00:32.913618+00:00"},{"alias_kind":"arxiv_version","alias_value":"1908.11365v1","created_at":"2026-07-05T00:00:32.913618+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1908.11365","created_at":"2026-07-05T00:00:32.913618+00:00"},{"alias_kind":"pith_short_12","alias_value":"EYSLPSUYN3UB","created_at":"2026-07-05T00:00:32.913618+00:00"},{"alias_kind":"pith_short_16","alias_value":"EYSLPSUYN3UBYVTQ","created_at":"2026-07-05T00:00:32.913618+00:00"},{"alias_kind":"pith_short_8","alias_value":"EYSLPSUY","created_at":"2026-07-05T00:00:32.913618+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2505.23013","citing_title":"Scalable Complexity Control Facilitates Reasoning Ability of LLMs","ref_index":78,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/EYSLPSUYN3UBYVTQ6M7QQTMY5E","json":"https://pith.science/pith/EYSLPSUYN3UBYVTQ6M7QQTMY5E.json","graph_json":"https://pith.science/api/pith-number/EYSLPSUYN3UBYVTQ6M7QQTMY5E/graph.json","events_json":"https://pith.science/api/pith-number/EYSLPSUYN3UBYVTQ6M7QQTMY5E/events.json","paper":"https://pith.science/paper/EYSLPSUY"},"agent_actions":{"view_html":"https://pith.science/pith/EYSLPSUYN3UBYVTQ6M7QQTMY5E","download_json":"https://pith.science/pith/EYSLPSUYN3UBYVTQ6M7QQTMY5E.json","view_paper":"https://pith.science/paper/EYSLPSUY","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=1908.11365&json=true","fetch_graph":"https://pith.science/api/pith-number/EYSLPSUYN3UBYVTQ6M7QQTMY5E/graph.json","fetch_events":"https://pith.science/api/pith-number/EYSLPSUYN3UBYVTQ6M7QQTMY5E/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/EYSLPSUYN3UBYVTQ6M7QQTMY5E/action/timestamp_anchor","attest_storage":"https://pith.science/pith/EYSLPSUYN3UBYVTQ6M7QQTMY5E/action/storage_attestation","attest_author":"https://pith.science/pith/EYSLPSUYN3UBYVTQ6M7QQTMY5E/action/author_attestation","sign_citation":"https://pith.science/pith/EYSLPSUYN3UBYVTQ6M7QQTMY5E/action/citation_signature","submit_replication":"https://pith.science/pith/EYSLPSUYN3UBYVTQ6M7QQTMY5E/action/replication_record"}},"created_at":"2026-07-05T00:00:32.913618+00:00","updated_at":"2026-07-05T00:00:32.913618+00:00"}