{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:W32W7T37XUZJXOKNZYADAS6NXD","short_pith_number":"pith:W32W7T37","schema_version":"1.0","canonical_sha256":"b6f56fcf7fbd329bb94dce00304bcdb8e4110d6c661af6a11bd53d31281db4be","source":{"kind":"arxiv","id":"2502.01563","version":4},"attestation_state":"computed","paper":{"title":"Massive Values in Self-Attention Modules are the Key to Contextual Knowledge Understanding","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Kai Mei, Mengnan Du, Mingjie Sun, Mingyu Jin, Ruixiang Tang, Wujiang Xu, Yongfeng Zhang, Zirui Liu","submitted_at":"2025-02-03T17:47:03Z","abstract_excerpt":"Large language models (LLMs) have achieved remarkable success in contextual knowledge understanding. In this paper, we show that these concentrated massive values consistently emerge in specific regions of attention queries (Q) and keys (K) while not having such patterns in values (V) in various modern transformer-based LLMs (Q, K, and V mean the representations output by the query, key, and value layers respectively). Through extensive experiments, we further demonstrate that these massive values play a critical role in interpreting contextual knowledge (knowledge obtained from the current co"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.01563","kind":"arxiv","version":4},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2025-02-03T17:47:03Z","cross_cats_sorted":[],"title_canon_sha256":"a52fce025e9a91e26b9de2a7d962d91ee89a75e216d1c1d09e64677df0060e2a","abstract_canon_sha256":"706695faa7399bc7e2d58f7199b44c1f580d7dba22a7477bbdb2d2e08446acbe"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:06:27.599613Z","signature_b64":"P73v/95M/+4xG5iMpBX3qeh94aUfYOV+TdHPuhdN3pqBjx2RNCKSnwb4AY8S+lhSD3UMDkqWgGsWOdchhI2tCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b6f56fcf7fbd329bb94dce00304bcdb8e4110d6c661af6a11bd53d31281db4be","last_reissued_at":"2026-07-05T11:06:27.599087Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:06:27.599087Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Massive Values in Self-Attention Modules are the Key to Contextual Knowledge Understanding","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Kai Mei, Mengnan Du, Mingjie Sun, Mingyu Jin, Ruixiang Tang, Wujiang Xu, Yongfeng Zhang, Zirui Liu","submitted_at":"2025-02-03T17:47:03Z","abstract_excerpt":"Large language models (LLMs) have achieved remarkable success in contextual knowledge understanding. In this paper, we show that these concentrated massive values consistently emerge in specific regions of attention queries (Q) and keys (K) while not having such patterns in values (V) in various modern transformer-based LLMs (Q, K, and V mean the representations output by the query, key, and value layers respectively). Through extensive experiments, we further demonstrate that these massive values play a critical role in interpreting contextual knowledge (knowledge obtained from the current co"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.01563","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.01563/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.01563","created_at":"2026-07-05T11:06:27.599148+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.01563v4","created_at":"2026-07-05T11:06:27.599148+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.01563","created_at":"2026-07-05T11:06:27.599148+00:00"},{"alias_kind":"pith_short_12","alias_value":"W32W7T37XUZJ","created_at":"2026-07-05T11:06:27.599148+00:00"},{"alias_kind":"pith_short_16","alias_value":"W32W7T37XUZJXOKN","created_at":"2026-07-05T11:06:27.599148+00:00"},{"alias_kind":"pith_short_8","alias_value":"W32W7T37","created_at":"2026-07-05T11:06:27.599148+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":8,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.22673","citing_title":"AgentLens: Interpretable Safety Steering via Mechanistic Subspaces for Multi-Turn Coding Agent","ref_index":62,"is_internal_anchor":false},{"citing_arxiv_id":"2605.19660","citing_title":"OScaR: The Occam's Razor for Extreme KV Cache Quantization in LLMs and Beyond","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10886","citing_title":"LoKA: Low-precision Kernel Applications for Recommendation Models At Scale","ref_index":41,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08504","citing_title":"A Single Layer to Explain Them All:Understanding Massive Activations in Large Language Models","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2410.02644","citing_title":"Agent Security Bench (ASB): Formalizing and Benchmarking Attacks and Defenses in LLM-based Agents","ref_index":109,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08504","citing_title":"A Single Layer to Explain Them All:Understanding Massive Activations in Large Language Models","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10886","citing_title":"LoKA: Low-precision Kernel Applications for Recommendation Models At Scale","ref_index":41,"is_internal_anchor":false},{"citing_arxiv_id":"2604.14922","citing_title":"LongAct: Harnessing Intrinsic Activation Patterns for Long-Context Reinforcement Learning","ref_index":15,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/W32W7T37XUZJXOKNZYADAS6NXD","json":"https://pith.science/pith/W32W7T37XUZJXOKNZYADAS6NXD.json","graph_json":"https://pith.science/api/pith-number/W32W7T37XUZJXOKNZYADAS6NXD/graph.json","events_json":"https://pith.science/api/pith-number/W32W7T37XUZJXOKNZYADAS6NXD/events.json","paper":"https://pith.science/paper/W32W7T37"},"agent_actions":{"view_html":"https://pith.science/pith/W32W7T37XUZJXOKNZYADAS6NXD","download_json":"https://pith.science/pith/W32W7T37XUZJXOKNZYADAS6NXD.json","view_paper":"https://pith.science/paper/W32W7T37","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.01563&json=true","fetch_graph":"https://pith.science/api/pith-number/W32W7T37XUZJXOKNZYADAS6NXD/graph.json","fetch_events":"https://pith.science/api/pith-number/W32W7T37XUZJXOKNZYADAS6NXD/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/W32W7T37XUZJXOKNZYADAS6NXD/action/timestamp_anchor","attest_storage":"https://pith.science/pith/W32W7T37XUZJXOKNZYADAS6NXD/action/storage_attestation","attest_author":"https://pith.science/pith/W32W7T37XUZJXOKNZYADAS6NXD/action/author_attestation","sign_citation":"https://pith.science/pith/W32W7T37XUZJXOKNZYADAS6NXD/action/citation_signature","submit_replication":"https://pith.science/pith/W32W7T37XUZJXOKNZYADAS6NXD/action/replication_record"}},"created_at":"2026-07-05T11:06:27.599148+00:00","updated_at":"2026-07-05T11:06:27.599148+00:00"}