{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:O6APZWS5SSBQ65LBMGAZLFXZAJ","short_pith_number":"pith:O6APZWS5","schema_version":"1.0","canonical_sha256":"7780fcda5d94830f756161819596f9026e253b06c1fd94909149140b6e17bab6","source":{"kind":"arxiv","id":"2306.12929","version":2},"attestation_state":"computed","paper":{"title":"Quantizable Transformers: Removing Outliers by Helping Attention Heads Do Nothing","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.CV"],"primary_cat":"cs.LG","authors_text":"Markus Nagel, Tijmen Blankevoort, Yelysei Bondarenko","submitted_at":"2023-06-22T14:39:04Z","abstract_excerpt":"Transformer models have been widely adopted in various domains over the last years, and especially large language models have advanced the field of AI significantly. Due to their size, the capability of these networks has increased tremendously, but this has come at the cost of a significant increase in necessary compute. Quantization is one of the most effective ways to reduce the computational time and memory consumption of neural networks. Many studies have shown, however, that modern transformer models tend to learn strong outliers in their activations, making them difficult to quantize. T"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2306.12929","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2023-06-22T14:39:04Z","cross_cats_sorted":["cs.AI","cs.CL","cs.CV"],"title_canon_sha256":"227fc47f4e4ed0484b6f7509c0b973df5256288f40eb6ea8e8a1c15cb19dbe47","abstract_canon_sha256":"31c9eaaa79a852ee528bcfbac177cf2f68e93e632463e66260fb95bd70009e45"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:10:48.607697Z","signature_b64":"knFkwqVwX8nCbS/G7i8aYD/jip1pDy19WGdcYhFXl+X97D0ij4xDsZirYtzq8vNr2m3zORd0GlhO7JzfaNDaBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"7780fcda5d94830f756161819596f9026e253b06c1fd94909149140b6e17bab6","last_reissued_at":"2026-07-05T07:10:48.606511Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:10:48.606511Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Quantizable Transformers: Removing Outliers by Helping Attention Heads Do Nothing","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.CV"],"primary_cat":"cs.LG","authors_text":"Markus Nagel, Tijmen Blankevoort, Yelysei Bondarenko","submitted_at":"2023-06-22T14:39:04Z","abstract_excerpt":"Transformer models have been widely adopted in various domains over the last years, and especially large language models have advanced the field of AI significantly. Due to their size, the capability of these networks has increased tremendously, but this has come at the cost of a significant increase in necessary compute. Quantization is one of the most effective ways to reduce the computational time and memory consumption of neural networks. Many studies have shown, however, that modern transformer models tend to learn strong outliers in their activations, making them difficult to quantize. T"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2306.12929","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2306.12929/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2306.12929","created_at":"2026-07-05T07:10:48.607135+00:00"},{"alias_kind":"arxiv_version","alias_value":"2306.12929v2","created_at":"2026-07-05T07:10:48.607135+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2306.12929","created_at":"2026-07-05T07:10:48.607135+00:00"},{"alias_kind":"pith_short_12","alias_value":"O6APZWS5SSBQ","created_at":"2026-07-05T07:10:48.607135+00:00"},{"alias_kind":"pith_short_16","alias_value":"O6APZWS5SSBQ65LB","created_at":"2026-07-05T07:10:48.607135+00:00"},{"alias_kind":"pith_short_8","alias_value":"O6APZWS5","created_at":"2026-07-05T07:10:48.607135+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":7,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.19005","citing_title":"Sumi: Open Uniform Diffusion Language Model from Scratch","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2606.20743","citing_title":"Massive Activations Are Architecturally Robust: A Controlled Scratch/Commitment Residual Stream Test","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2606.07604","citing_title":"Contribution Weights: A Geometrical Analysis of Self-Attention Transformers","ref_index":146,"is_internal_anchor":false},{"citing_arxiv_id":"2606.07604","citing_title":"Contribution Weights: A Geometrical Analysis of Self-Attention Transformers","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2512.02010","citing_title":"Four Over Six: More Accurate NVFP4 Quantization with Adaptive Block Scaling","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2602.01203","citing_title":"Attention Sink Forges Native MoE in Attention Layers: Sink-Aware Training to Address Head Collapse","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2402.17762","citing_title":"Massive Activations in Large Language Models","ref_index":108,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/O6APZWS5SSBQ65LBMGAZLFXZAJ","json":"https://pith.science/pith/O6APZWS5SSBQ65LBMGAZLFXZAJ.json","graph_json":"https://pith.science/api/pith-number/O6APZWS5SSBQ65LBMGAZLFXZAJ/graph.json","events_json":"https://pith.science/api/pith-number/O6APZWS5SSBQ65LBMGAZLFXZAJ/events.json","paper":"https://pith.science/paper/O6APZWS5"},"agent_actions":{"view_html":"https://pith.science/pith/O6APZWS5SSBQ65LBMGAZLFXZAJ","download_json":"https://pith.science/pith/O6APZWS5SSBQ65LBMGAZLFXZAJ.json","view_paper":"https://pith.science/paper/O6APZWS5","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2306.12929&json=true","fetch_graph":"https://pith.science/api/pith-number/O6APZWS5SSBQ65LBMGAZLFXZAJ/graph.json","fetch_events":"https://pith.science/api/pith-number/O6APZWS5SSBQ65LBMGAZLFXZAJ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/O6APZWS5SSBQ65LBMGAZLFXZAJ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/O6APZWS5SSBQ65LBMGAZLFXZAJ/action/storage_attestation","attest_author":"https://pith.science/pith/O6APZWS5SSBQ65LBMGAZLFXZAJ/action/author_attestation","sign_citation":"https://pith.science/pith/O6APZWS5SSBQ65LBMGAZLFXZAJ/action/citation_signature","submit_replication":"https://pith.science/pith/O6APZWS5SSBQ65LBMGAZLFXZAJ/action/replication_record"}},"created_at":"2026-07-05T07:10:48.607135+00:00","updated_at":"2026-07-05T07:10:48.607135+00:00"}