{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:DXJZU6SFALULL4BECI736Y66AG","short_pith_number":"pith:DXJZU6SF","schema_version":"1.0","canonical_sha256":"1dd39a7a4502e8b5f024123fbf63de0190ec56f253d7e34ca883401c77d3a0ec","source":{"kind":"arxiv","id":"2109.04404","version":1},"attestation_state":"computed","paper":{"title":"All Bark and No Bite: Rogue Dimensions in Transformer Language Models Obscure Representational Quality","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Marten van Schijndel, William Timkey","submitted_at":"2021-09-09T16:45:15Z","abstract_excerpt":"Similarity measures are a vital tool for understanding how language models represent and process language. Standard representational similarity measures such as cosine similarity and Euclidean distance have been successfully used in static word embedding models to understand how words cluster in semantic space. Recently, these measures have been applied to embeddings from contextualized models such as BERT and GPT-2. In this work, we call into question the informativity of such measures for contextualized language models. We find that a small number of rogue dimensions, often just 1-3, dominat"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2109.04404","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2021-09-09T16:45:15Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"c5a6b75d8d11e8c2bb4d9ddf9462963a25e4d5737041fd63c3ab69168ed8969a","abstract_canon_sha256":"d5efe1f33f2d1146386410302d41139ab9d0171aa7f84f6f80201ac45a54d006"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T03:12:58.162279Z","signature_b64":"r8V3zTitGMQWkcjuK8mv0X4tUGaCL686aMcj3MHiXQwVWcl9hV27DTJ6m0peT2rVje1s83AbRcG19KZwJjNiAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"1dd39a7a4502e8b5f024123fbf63de0190ec56f253d7e34ca883401c77d3a0ec","last_reissued_at":"2026-07-05T03:12:58.161878Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T03:12:58.161878Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"All Bark and No Bite: Rogue Dimensions in Transformer Language Models Obscure Representational Quality","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Marten van Schijndel, William Timkey","submitted_at":"2021-09-09T16:45:15Z","abstract_excerpt":"Similarity measures are a vital tool for understanding how language models represent and process language. Standard representational similarity measures such as cosine similarity and Euclidean distance have been successfully used in static word embedding models to understand how words cluster in semantic space. Recently, these measures have been applied to embeddings from contextualized models such as BERT and GPT-2. In this work, we call into question the informativity of such measures for contextualized language models. We find that a small number of rogue dimensions, often just 1-3, dominat"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2109.04404","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2109.04404/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2109.04404","created_at":"2026-07-05T03:12:58.161930+00:00"},{"alias_kind":"arxiv_version","alias_value":"2109.04404v1","created_at":"2026-07-05T03:12:58.161930+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2109.04404","created_at":"2026-07-05T03:12:58.161930+00:00"},{"alias_kind":"pith_short_12","alias_value":"DXJZU6SFALUL","created_at":"2026-07-05T03:12:58.161930+00:00"},{"alias_kind":"pith_short_16","alias_value":"DXJZU6SFALULL4BE","created_at":"2026-07-05T03:12:58.161930+00:00"},{"alias_kind":"pith_short_8","alias_value":"DXJZU6SF","created_at":"2026-07-05T03:12:58.161930+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":8,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.20743","citing_title":"Massive Activations Are Architecturally Robust: A Controlled Scratch/Commitment Residual Stream Test","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2606.15054","citing_title":"Size Doesn't Matter: Cosine-Scored Sparse Autoencoders","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2402.17762","citing_title":"Massive Activations in Large Language Models","ref_index":150,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08504","citing_title":"A Single Layer to Explain Them All:Understanding Massive Activations in Large Language Models","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2208.07339","citing_title":"LLM.int8(): 8-bit Matrix Multiplication for Transformers at Scale","ref_index":163,"is_internal_anchor":false},{"citing_arxiv_id":"2303.08112","citing_title":"Eliciting Latent Predictions from Transformers with the Tuned Lens","ref_index":85,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08504","citing_title":"A Single Layer to Explain Them All:Understanding Massive Activations in Large Language Models","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05741","citing_title":"HyperLens: Quantifying Cognitive Effort in LLMs with Fine-grained Confidence Trajectory","ref_index":29,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/DXJZU6SFALULL4BECI736Y66AG","json":"https://pith.science/pith/DXJZU6SFALULL4BECI736Y66AG.json","graph_json":"https://pith.science/api/pith-number/DXJZU6SFALULL4BECI736Y66AG/graph.json","events_json":"https://pith.science/api/pith-number/DXJZU6SFALULL4BECI736Y66AG/events.json","paper":"https://pith.science/paper/DXJZU6SF"},"agent_actions":{"view_html":"https://pith.science/pith/DXJZU6SFALULL4BECI736Y66AG","download_json":"https://pith.science/pith/DXJZU6SFALULL4BECI736Y66AG.json","view_paper":"https://pith.science/paper/DXJZU6SF","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2109.04404&json=true","fetch_graph":"https://pith.science/api/pith-number/DXJZU6SFALULL4BECI736Y66AG/graph.json","fetch_events":"https://pith.science/api/pith-number/DXJZU6SFALULL4BECI736Y66AG/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/DXJZU6SFALULL4BECI736Y66AG/action/timestamp_anchor","attest_storage":"https://pith.science/pith/DXJZU6SFALULL4BECI736Y66AG/action/storage_attestation","attest_author":"https://pith.science/pith/DXJZU6SFALULL4BECI736Y66AG/action/author_attestation","sign_citation":"https://pith.science/pith/DXJZU6SFALULL4BECI736Y66AG/action/citation_signature","submit_replication":"https://pith.science/pith/DXJZU6SFALULL4BECI736Y66AG/action/replication_record"}},"created_at":"2026-07-05T03:12:58.161930+00:00","updated_at":"2026-07-05T03:12:58.161930+00:00"}