{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:MZTO45HEYUZX6EIYYGAM2FXK2A","short_pith_number":"pith:MZTO45HE","schema_version":"1.0","canonical_sha256":"6666ee74e4c5337f1118c180cd16ead03999e40c1aa9ad89d68f03a9db4c2a25","source":{"kind":"arxiv","id":"2410.01131","version":2},"attestation_state":"computed","paper":{"title":"nGPT: Normalized Transformer with Representation Learning on the Hypersphere","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Boris Ginsburg, Cheng-Ping Hsieh, Ilya Loshchilov, Simeng Sun","submitted_at":"2024-10-01T23:50:09Z","abstract_excerpt":"We propose a novel neural network architecture, the normalized Transformer (nGPT) with representation learning on the hypersphere. In nGPT, all vectors forming the embeddings, MLP, attention matrices and hidden states are unit norm normalized. The input stream of tokens travels on the surface of a hypersphere, with each layer contributing a displacement towards the target output predictions. These displacements are defined by the MLP and attention blocks, whose vector components also reside on the same hypersphere. Experiments show that nGPT learns much faster, reducing the number of training "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.01131","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-10-01T23:50:09Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"4b9b8cb849f59ec2813a83ca9d1d178c6b8afc98e7adf591fd0b81384fcf7f70","abstract_canon_sha256":"77c8d92336b699aa4553ff2e3b2139de6bf32be2767ccf048da79f710547b1ab"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:53:12.060529Z","signature_b64":"MCSfLDTDRzo3MB45V+DRS4WabT8ImI/rKJofi8eJieyzHOo1cwXEbHuDEAgFoJL2p3AoyeHmEb+cpZdmpXYiDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"6666ee74e4c5337f1118c180cd16ead03999e40c1aa9ad89d68f03a9db4c2a25","last_reissued_at":"2026-07-05T10:53:12.060058Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:53:12.060058Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"nGPT: Normalized Transformer with Representation Learning on the Hypersphere","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Boris Ginsburg, Cheng-Ping Hsieh, Ilya Loshchilov, Simeng Sun","submitted_at":"2024-10-01T23:50:09Z","abstract_excerpt":"We propose a novel neural network architecture, the normalized Transformer (nGPT) with representation learning on the hypersphere. In nGPT, all vectors forming the embeddings, MLP, attention matrices and hidden states are unit norm normalized. The input stream of tokens travels on the surface of a hypersphere, with each layer contributing a displacement towards the target output predictions. These displacements are defined by the MLP and attention blocks, whose vector components also reside on the same hypersphere. Experiments show that nGPT learns much faster, reducing the number of training "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.01131","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.01131/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.01131","created_at":"2026-07-05T10:53:12.060113+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.01131v2","created_at":"2026-07-05T10:53:12.060113+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.01131","created_at":"2026-07-05T10:53:12.060113+00:00"},{"alias_kind":"pith_short_12","alias_value":"MZTO45HEYUZX","created_at":"2026-07-05T10:53:12.060113+00:00"},{"alias_kind":"pith_short_16","alias_value":"MZTO45HEYUZX6EIY","created_at":"2026-07-05T10:53:12.060113+00:00"},{"alias_kind":"pith_short_8","alias_value":"MZTO45HE","created_at":"2026-07-05T10:53:12.060113+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":13,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.25971","citing_title":"Improving Neural Network Training by Decoupling the Magnitude and Direction of Weight Vectors","ref_index":100,"is_internal_anchor":false},{"citing_arxiv_id":"2606.15054","citing_title":"Size Doesn't Matter: Cosine-Scored Sparse Autoencoders","ref_index":48,"is_internal_anchor":false},{"citing_arxiv_id":"2503.17715","citing_title":"Normalized Matching Transformer","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2604.04539","citing_title":"FlashSAC: Fast and Stable Off-Policy Reinforcement Learning for High-Dimensional Robot Control","ref_index":46,"is_internal_anchor":false},{"citing_arxiv_id":"2509.04154","citing_title":"Robust Filter Attention: Self-Attention as Precision-Weighted State Estimation","ref_index":56,"is_internal_anchor":false},{"citing_arxiv_id":"2605.13262","citing_title":"Chem-GMNet: A Sphere-Native Geometric Transformer for Molecular Property Prediction","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11850","citing_title":"Constrained Stochastic Spectral Preconditioning Converges for Nonconvex Objectives","ref_index":43,"is_internal_anchor":false},{"citing_arxiv_id":"2604.23434","citing_title":"When Does Removing LayerNorm Help? Activation Bounding as a Regime-Dependent Implicit Regularizer","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06611","citing_title":"The Structural Origin of Attention Sink: Variance Discrepancy, Super Neurons, and Dimension Disparity","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2605.04418","citing_title":"Demystifying Manifold Constraints in LLM Pre-training","ref_index":47,"is_internal_anchor":false},{"citing_arxiv_id":"2605.00265","citing_title":"Polaris: Coupled Orbital Polar Embeddings for Hierarchical Concept Learning","ref_index":61,"is_internal_anchor":false},{"citing_arxiv_id":"2604.04539","citing_title":"FlashSAC: Fast and Stable Off-Policy Reinforcement Learning for High-Dimensional Robot Control","ref_index":46,"is_internal_anchor":false},{"citing_arxiv_id":"2505.10465","citing_title":"Superposition Yields Robust Neural Scaling","ref_index":60,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/MZTO45HEYUZX6EIYYGAM2FXK2A","json":"https://pith.science/pith/MZTO45HEYUZX6EIYYGAM2FXK2A.json","graph_json":"https://pith.science/api/pith-number/MZTO45HEYUZX6EIYYGAM2FXK2A/graph.json","events_json":"https://pith.science/api/pith-number/MZTO45HEYUZX6EIYYGAM2FXK2A/events.json","paper":"https://pith.science/paper/MZTO45HE"},"agent_actions":{"view_html":"https://pith.science/pith/MZTO45HEYUZX6EIYYGAM2FXK2A","download_json":"https://pith.science/pith/MZTO45HEYUZX6EIYYGAM2FXK2A.json","view_paper":"https://pith.science/paper/MZTO45HE","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.01131&json=true","fetch_graph":"https://pith.science/api/pith-number/MZTO45HEYUZX6EIYYGAM2FXK2A/graph.json","fetch_events":"https://pith.science/api/pith-number/MZTO45HEYUZX6EIYYGAM2FXK2A/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/MZTO45HEYUZX6EIYYGAM2FXK2A/action/timestamp_anchor","attest_storage":"https://pith.science/pith/MZTO45HEYUZX6EIYYGAM2FXK2A/action/storage_attestation","attest_author":"https://pith.science/pith/MZTO45HEYUZX6EIYYGAM2FXK2A/action/author_attestation","sign_citation":"https://pith.science/pith/MZTO45HEYUZX6EIYYGAM2FXK2A/action/citation_signature","submit_replication":"https://pith.science/pith/MZTO45HEYUZX6EIYYGAM2FXK2A/action/replication_record"}},"created_at":"2026-07-05T10:53:12.060113+00:00","updated_at":"2026-07-05T10:53:12.060113+00:00"}