{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2019:FWPECXTXVJISALNJPMX3BB3ATK","short_pith_number":"pith:FWPECXTX","schema_version":"1.0","canonical_sha256":"2d9e415e77aa51202da97b2fb087609a92191dd1a3d36c96999ce0947263549c","source":{"kind":"arxiv","id":"1911.07013","version":1},"attestation_state":"computed","paper":{"title":"Understanding and Improving Layer Normalization","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL","stat.ML"],"primary_cat":"cs.LG","authors_text":"Guangxiang Zhao, Jingjing Xu, Junyang Lin, Xu Sun, Zhiyuan Zhang","submitted_at":"2019-11-16T11:00:16Z","abstract_excerpt":"Layer normalization (LayerNorm) is a technique to normalize the distributions of intermediate layers. It enables smoother gradients, faster training, and better generalization accuracy. However, it is still unclear where the effectiveness stems from. In this paper, our main contribution is to take a step further in understanding LayerNorm. Many of previous studies believe that the success of LayerNorm comes from forward normalization. Unlike them, we find that the derivatives of the mean and variance are more important than forward normalization by re-centering and re-scaling backward gradient"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"1911.07013","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2019-11-16T11:00:16Z","cross_cats_sorted":["cs.CL","stat.ML"],"title_canon_sha256":"1bdf3e3d6d38899e5d2d8f35a63b972c3891436b5958f104bba1b196548c3f2a","abstract_canon_sha256":"b7f6383c409c0a763b530f9ba682d7d5aac2123c382a0b726d8575f439b24376"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T00:20:03.056678Z","signature_b64":"c0+J98Xb/CqCq6SX69We7P0UwTJ3yoGzt7ZFiIYVNVLrobDv7AVgDspCvff2murJDVJHpsYA9RFV0Q75X4ViBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"2d9e415e77aa51202da97b2fb087609a92191dd1a3d36c96999ce0947263549c","last_reissued_at":"2026-07-05T00:20:03.056175Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T00:20:03.056175Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Understanding and Improving Layer Normalization","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL","stat.ML"],"primary_cat":"cs.LG","authors_text":"Guangxiang Zhao, Jingjing Xu, Junyang Lin, Xu Sun, Zhiyuan Zhang","submitted_at":"2019-11-16T11:00:16Z","abstract_excerpt":"Layer normalization (LayerNorm) is a technique to normalize the distributions of intermediate layers. It enables smoother gradients, faster training, and better generalization accuracy. However, it is still unclear where the effectiveness stems from. In this paper, our main contribution is to take a step further in understanding LayerNorm. Many of previous studies believe that the success of LayerNorm comes from forward normalization. Unlike them, we find that the derivatives of the mean and variance are more important than forward normalization by re-centering and re-scaling backward gradient"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1911.07013","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/1911.07013/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"1911.07013","created_at":"2026-07-05T00:20:03.056232+00:00"},{"alias_kind":"arxiv_version","alias_value":"1911.07013v1","created_at":"2026-07-05T00:20:03.056232+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1911.07013","created_at":"2026-07-05T00:20:03.056232+00:00"},{"alias_kind":"pith_short_12","alias_value":"FWPECXTXVJIS","created_at":"2026-07-05T00:20:03.056232+00:00"},{"alias_kind":"pith_short_16","alias_value":"FWPECXTXVJISALNJ","created_at":"2026-07-05T00:20:03.056232+00:00"},{"alias_kind":"pith_short_8","alias_value":"FWPECXTX","created_at":"2026-07-05T00:20:03.056232+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.06591","citing_title":"BRICKS: Compositional Neural Markov Kernels for Zero-Shot Radiation-Matter Simulation","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2604.12946","citing_title":"Parcae: Scaling Laws For Stable Looped Language Models","ref_index":86,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/FWPECXTXVJISALNJPMX3BB3ATK","json":"https://pith.science/pith/FWPECXTXVJISALNJPMX3BB3ATK.json","graph_json":"https://pith.science/api/pith-number/FWPECXTXVJISALNJPMX3BB3ATK/graph.json","events_json":"https://pith.science/api/pith-number/FWPECXTXVJISALNJPMX3BB3ATK/events.json","paper":"https://pith.science/paper/FWPECXTX"},"agent_actions":{"view_html":"https://pith.science/pith/FWPECXTXVJISALNJPMX3BB3ATK","download_json":"https://pith.science/pith/FWPECXTXVJISALNJPMX3BB3ATK.json","view_paper":"https://pith.science/paper/FWPECXTX","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=1911.07013&json=true","fetch_graph":"https://pith.science/api/pith-number/FWPECXTXVJISALNJPMX3BB3ATK/graph.json","fetch_events":"https://pith.science/api/pith-number/FWPECXTXVJISALNJPMX3BB3ATK/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/FWPECXTXVJISALNJPMX3BB3ATK/action/timestamp_anchor","attest_storage":"https://pith.science/pith/FWPECXTXVJISALNJPMX3BB3ATK/action/storage_attestation","attest_author":"https://pith.science/pith/FWPECXTXVJISALNJPMX3BB3ATK/action/author_attestation","sign_citation":"https://pith.science/pith/FWPECXTXVJISALNJPMX3BB3ATK/action/citation_signature","submit_replication":"https://pith.science/pith/FWPECXTXVJISALNJPMX3BB3ATK/action/replication_record"}},"created_at":"2026-07-05T00:20:03.056232+00:00","updated_at":"2026-07-05T00:20:03.056232+00:00"}