{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:GQRKJTF4GH3IFAOEZJPYJXCHPB","short_pith_number":"pith:GQRKJTF4","schema_version":"1.0","canonical_sha256":"3422a4ccbc31f68281c4ca5f84dc4778624d08b2d7239ea242ebdae958dbcf09","source":{"kind":"arxiv","id":"2305.14858","version":2},"attestation_state":"computed","paper":{"title":"Pre-RMSNorm and Pre-CRMSNorm Transformers: Equivalent and Efficient Pre-LN Transformers","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.NE"],"primary_cat":"cs.LG","authors_text":"David Z. Pan, Hanqing Zhu, Jiaqi Gu, Zixuan Jiang","submitted_at":"2023-05-24T08:08:26Z","abstract_excerpt":"Transformers have achieved great success in machine learning applications. Normalization techniques, such as Layer Normalization (LayerNorm, LN) and Root Mean Square Normalization (RMSNorm), play a critical role in accelerating and stabilizing the training of Transformers. While LayerNorm recenters and rescales input vectors, RMSNorm only rescales the vectors by their RMS value. Despite being more computationally efficient, RMSNorm may compromise the representation ability of Transformers. There is currently no consensus regarding the preferred normalization technique, as some models employ La"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2305.14858","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2023-05-24T08:08:26Z","cross_cats_sorted":["cs.AI","cs.NE"],"title_canon_sha256":"fd8c2884d0a1eb9d4084b5eaf73e4df3541ad8fc496e7656f5cf976d18dc712f","abstract_canon_sha256":"61231e449a0e84582dcbc4e6d262079c0b81d26f70e241a1d74f4d48d44e5e32"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:05:09.058307Z","signature_b64":"dfMjyLbJEgyxgxZ/+Okj5M4eeoTZLiALuveSjKlJ7IX99wQH9wxsStwyrT9axznmJPSvuMq2vmEwV9ne1F6uBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"3422a4ccbc31f68281c4ca5f84dc4778624d08b2d7239ea242ebdae958dbcf09","last_reissued_at":"2026-07-05T07:05:09.057820Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:05:09.057820Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Pre-RMSNorm and Pre-CRMSNorm Transformers: Equivalent and Efficient Pre-LN Transformers","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.NE"],"primary_cat":"cs.LG","authors_text":"David Z. Pan, Hanqing Zhu, Jiaqi Gu, Zixuan Jiang","submitted_at":"2023-05-24T08:08:26Z","abstract_excerpt":"Transformers have achieved great success in machine learning applications. Normalization techniques, such as Layer Normalization (LayerNorm, LN) and Root Mean Square Normalization (RMSNorm), play a critical role in accelerating and stabilizing the training of Transformers. While LayerNorm recenters and rescales input vectors, RMSNorm only rescales the vectors by their RMS value. Despite being more computationally efficient, RMSNorm may compromise the representation ability of Transformers. There is currently no consensus regarding the preferred normalization technique, as some models employ La"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2305.14858","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2305.14858/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2305.14858","created_at":"2026-07-05T07:05:09.057879+00:00"},{"alias_kind":"arxiv_version","alias_value":"2305.14858v2","created_at":"2026-07-05T07:05:09.057879+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2305.14858","created_at":"2026-07-05T07:05:09.057879+00:00"},{"alias_kind":"pith_short_12","alias_value":"GQRKJTF4GH3I","created_at":"2026-07-05T07:05:09.057879+00:00"},{"alias_kind":"pith_short_16","alias_value":"GQRKJTF4GH3IFAOE","created_at":"2026-07-05T07:05:09.057879+00:00"},{"alias_kind":"pith_short_8","alias_value":"GQRKJTF4","created_at":"2026-07-05T07:05:09.057879+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.23791","citing_title":"One Generator, Any Process: LLM-Conditioning for the LHC","ref_index":254,"is_internal_anchor":false},{"citing_arxiv_id":"2606.23791","citing_title":"One Generator, Any Process: LLM-Conditioning for the LHC","ref_index":258,"is_internal_anchor":false},{"citing_arxiv_id":"2309.16609","citing_title":"Qwen Technical Report","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2505.09388","citing_title":"Qwen3 Technical Report","ref_index":21,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/GQRKJTF4GH3IFAOEZJPYJXCHPB","json":"https://pith.science/pith/GQRKJTF4GH3IFAOEZJPYJXCHPB.json","graph_json":"https://pith.science/api/pith-number/GQRKJTF4GH3IFAOEZJPYJXCHPB/graph.json","events_json":"https://pith.science/api/pith-number/GQRKJTF4GH3IFAOEZJPYJXCHPB/events.json","paper":"https://pith.science/paper/GQRKJTF4"},"agent_actions":{"view_html":"https://pith.science/pith/GQRKJTF4GH3IFAOEZJPYJXCHPB","download_json":"https://pith.science/pith/GQRKJTF4GH3IFAOEZJPYJXCHPB.json","view_paper":"https://pith.science/paper/GQRKJTF4","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2305.14858&json=true","fetch_graph":"https://pith.science/api/pith-number/GQRKJTF4GH3IFAOEZJPYJXCHPB/graph.json","fetch_events":"https://pith.science/api/pith-number/GQRKJTF4GH3IFAOEZJPYJXCHPB/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/GQRKJTF4GH3IFAOEZJPYJXCHPB/action/timestamp_anchor","attest_storage":"https://pith.science/pith/GQRKJTF4GH3IFAOEZJPYJXCHPB/action/storage_attestation","attest_author":"https://pith.science/pith/GQRKJTF4GH3IFAOEZJPYJXCHPB/action/author_attestation","sign_citation":"https://pith.science/pith/GQRKJTF4GH3IFAOEZJPYJXCHPB/action/citation_signature","submit_replication":"https://pith.science/pith/GQRKJTF4GH3IFAOEZJPYJXCHPB/action/replication_record"}},"created_at":"2026-07-05T07:05:09.057879+00:00","updated_at":"2026-07-05T07:05:09.057879+00:00"}