{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:J36PKNAAJMEO76NERSKPX6Q2L5","short_pith_number":"pith:J36PKNAA","schema_version":"1.0","canonical_sha256":"4efcf534004b08eff9a48c94fbfa1a5f576e0993d51d95a498efdd55d1970b5a","source":{"kind":"arxiv","id":"2503.22329","version":1},"attestation_state":"computed","paper":{"title":"A Refined Analysis of Massive Activations in LLMs","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Abhay Kumar, Fabian G\\\"ura, Louis Owen, Nilabhra Roy Chowdhury","submitted_at":"2025-03-28T11:08:34Z","abstract_excerpt":"Motivated in part by their relevance for low-precision training and quantization, massive activations in large language models (LLMs) have recently emerged as a topic of interest. However, existing analyses are limited in scope, and generalizability across architectures is unclear. This paper helps address some of these gaps by conducting an analysis of massive activations across a broad range of LLMs, including both GLU-based and non-GLU-based architectures. Our findings challenge several prior assumptions, most importantly: (1) not all massive activations are detrimental, i.e. suppressing th"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2503.22329","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-03-28T11:08:34Z","cross_cats_sorted":[],"title_canon_sha256":"9394b01de38a0c0e61b75aa68a27c9b4aae638afb04087e6a48fd7fdb18b9888","abstract_canon_sha256":"96ef10d1c6fc4f0f419d9ed1272371b0e705cc6ea661205571465ab93a83b606"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:40:52.552150Z","signature_b64":"3fkO0Hp241zIeUzQalwo83e8OAD4Lcsa7ns7aYY1VDEAUp/jrFHLuClE6TH+aPNbRJhGY+uiYw3kdSEYg9o4BA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"4efcf534004b08eff9a48c94fbfa1a5f576e0993d51d95a498efdd55d1970b5a","last_reissued_at":"2026-07-05T10:40:52.551485Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:40:52.551485Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"A Refined Analysis of Massive Activations in LLMs","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Abhay Kumar, Fabian G\\\"ura, Louis Owen, Nilabhra Roy Chowdhury","submitted_at":"2025-03-28T11:08:34Z","abstract_excerpt":"Motivated in part by their relevance for low-precision training and quantization, massive activations in large language models (LLMs) have recently emerged as a topic of interest. However, existing analyses are limited in scope, and generalizability across architectures is unclear. This paper helps address some of these gaps by conducting an analysis of massive activations across a broad range of LLMs, including both GLU-based and non-GLU-based architectures. Our findings challenge several prior assumptions, most importantly: (1) not all massive activations are detrimental, i.e. suppressing th"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2503.22329","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2503.22329/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2503.22329","created_at":"2026-07-05T10:40:52.551566+00:00"},{"alias_kind":"arxiv_version","alias_value":"2503.22329v1","created_at":"2026-07-05T10:40:52.551566+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2503.22329","created_at":"2026-07-05T10:40:52.551566+00:00"},{"alias_kind":"pith_short_12","alias_value":"J36PKNAAJMEO","created_at":"2026-07-05T10:40:52.551566+00:00"},{"alias_kind":"pith_short_16","alias_value":"J36PKNAAJMEO76NE","created_at":"2026-07-05T10:40:52.551566+00:00"},{"alias_kind":"pith_short_8","alias_value":"J36PKNAA","created_at":"2026-07-05T10:40:52.551566+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2603.17771","citing_title":"Attention Sinks Induce Gradient Sinks: Massive Activations as Gradient Regulators in Transformers","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08504","citing_title":"A Single Layer to Explain Them All:Understanding Massive Activations in Large Language Models","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08504","citing_title":"A Single Layer to Explain Them All:Understanding Massive Activations in Large Language Models","ref_index":12,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/J36PKNAAJMEO76NERSKPX6Q2L5","json":"https://pith.science/pith/J36PKNAAJMEO76NERSKPX6Q2L5.json","graph_json":"https://pith.science/api/pith-number/J36PKNAAJMEO76NERSKPX6Q2L5/graph.json","events_json":"https://pith.science/api/pith-number/J36PKNAAJMEO76NERSKPX6Q2L5/events.json","paper":"https://pith.science/paper/J36PKNAA"},"agent_actions":{"view_html":"https://pith.science/pith/J36PKNAAJMEO76NERSKPX6Q2L5","download_json":"https://pith.science/pith/J36PKNAAJMEO76NERSKPX6Q2L5.json","view_paper":"https://pith.science/paper/J36PKNAA","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2503.22329&json=true","fetch_graph":"https://pith.science/api/pith-number/J36PKNAAJMEO76NERSKPX6Q2L5/graph.json","fetch_events":"https://pith.science/api/pith-number/J36PKNAAJMEO76NERSKPX6Q2L5/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/J36PKNAAJMEO76NERSKPX6Q2L5/action/timestamp_anchor","attest_storage":"https://pith.science/pith/J36PKNAAJMEO76NERSKPX6Q2L5/action/storage_attestation","attest_author":"https://pith.science/pith/J36PKNAAJMEO76NERSKPX6Q2L5/action/author_attestation","sign_citation":"https://pith.science/pith/J36PKNAAJMEO76NERSKPX6Q2L5/action/citation_signature","submit_replication":"https://pith.science/pith/J36PKNAAJMEO76NERSKPX6Q2L5/action/replication_record"}},"created_at":"2026-07-05T10:40:52.551566+00:00","updated_at":"2026-07-05T10:40:52.551566+00:00"}