{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:PRTWLIKRP2RBFQQEE6UKUWY6OE","short_pith_number":"pith:PRTWLIKR","schema_version":"1.0","canonical_sha256":"7c6765a1517ea212c20427a8aa5b1e711137e075fd0ecccbf6b0899ce5d83fe8","source":{"kind":"arxiv","id":"2210.01892","version":4},"attestation_state":"computed","paper":{"title":"Polysemanticity and Capacity in Neural Networks","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.NE","authors_text":"Adam Scherlis, Adam S. Jermyn, Buck Shlegeris, Joe Benton, Kshitij Sachan","submitted_at":"2022-10-04T20:28:43Z","abstract_excerpt":"Individual neurons in neural networks often represent a mixture of unrelated features. This phenomenon, called polysemanticity, can make interpreting neural networks more difficult and so we aim to understand its causes. We propose doing so through the lens of feature \\emph{capacity}, which is the fractional dimension each feature consumes in the embedding space. We show that in a toy model the optimal capacity allocation tends to monosemantically represent the most important features, polysemantically represent less important features (in proportion to their impact on the loss), and entirely "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2210.01892","kind":"arxiv","version":4},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.NE","submitted_at":"2022-10-04T20:28:43Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"f1afcf76bd5a356e8e925486d01339860f04be153c48df6186e2fc60d24ed866","abstract_canon_sha256":"d8e723d3cab09b1070cb95b0bfaa4938a1f333a595d18ecd6b59440e22ca6d36"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:38:29.875335Z","signature_b64":"A6bOjt4mjbYdUtwAkK8zQORXZi53X6NiNsrAo1QxBuxc/7qmkCqN/GaUdB8TPboXzZhh085XMo1sXIBmwVmXCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"7c6765a1517ea212c20427a8aa5b1e711137e075fd0ecccbf6b0899ce5d83fe8","last_reissued_at":"2026-07-05T10:38:29.874736Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:38:29.874736Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Polysemanticity and Capacity in Neural Networks","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.NE","authors_text":"Adam Scherlis, Adam S. Jermyn, Buck Shlegeris, Joe Benton, Kshitij Sachan","submitted_at":"2022-10-04T20:28:43Z","abstract_excerpt":"Individual neurons in neural networks often represent a mixture of unrelated features. This phenomenon, called polysemanticity, can make interpreting neural networks more difficult and so we aim to understand its causes. We propose doing so through the lens of feature \\emph{capacity}, which is the fractional dimension each feature consumes in the embedding space. We show that in a toy model the optimal capacity allocation tends to monosemantically represent the most important features, polysemantically represent less important features (in proportion to their impact on the loss), and entirely "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2210.01892","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2210.01892/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2210.01892","created_at":"2026-07-05T10:38:29.874800+00:00"},{"alias_kind":"arxiv_version","alias_value":"2210.01892v4","created_at":"2026-07-05T10:38:29.874800+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2210.01892","created_at":"2026-07-05T10:38:29.874800+00:00"},{"alias_kind":"pith_short_12","alias_value":"PRTWLIKRP2RB","created_at":"2026-07-05T10:38:29.874800+00:00"},{"alias_kind":"pith_short_16","alias_value":"PRTWLIKRP2RBFQQE","created_at":"2026-07-05T10:38:29.874800+00:00"},{"alias_kind":"pith_short_8","alias_value":"PRTWLIKR","created_at":"2026-07-05T10:38:29.874800+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":18,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.26629","citing_title":"From Weights to Features: SAE-Guided Activation Regularization for LLM Continual Learning","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2606.18897","citing_title":"SAERec: Constructing Fine-grained Interpretable Intents Priors via Sparse Autoencoders for Recommendation","ref_index":39,"is_internal_anchor":false},{"citing_arxiv_id":"2606.03990","citing_title":"Neuron Populations Exhibit Divergent Selectivity with Scale","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2606.27440","citing_title":"PairSAE: Mechanistic Interpretability from Pair Representations in Protein Co-Folding","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2605.29358","citing_title":"Scaling Monosemanticity: Extracting Interpretable Features from Claude 3 Sonnet","ref_index":65,"is_internal_anchor":false},{"citing_arxiv_id":"2605.31304","citing_title":"Interpretability Without Tradeoffs: Disentangling Polysemanticity At Equal Predictive Performance","ref_index":56,"is_internal_anchor":false},{"citing_arxiv_id":"2605.23040","citing_title":"Steered Generation via Gradient-Based Optimization on Sparse Query Features","ref_index":37,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12770","citing_title":"WriteSAE: Sparse Autoencoders for Recurrent State","ref_index":37,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12770","citing_title":"WriteSAE: Sparse Autoencoders for Recurrent State","ref_index":91,"is_internal_anchor":false},{"citing_arxiv_id":"2309.16042","citing_title":"Towards Best Practices of Activation Patching in Language Models: Metrics and Methods","ref_index":110,"is_internal_anchor":false},{"citing_arxiv_id":"2511.20779","citing_title":"CHiQPM: Calibrated Hierarchical Interpretable Image Classification","ref_index":54,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12770","citing_title":"WriteSAE: Sparse Autoencoders for Recurrent State","ref_index":91,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12770","citing_title":"WriteSAE: Sparse Autoencoders for Recurrent State","ref_index":91,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12809","citing_title":"Correcting Influence: Unboxing LLM Outputs with Orthogonal Latent Spaces","ref_index":157,"is_internal_anchor":false},{"citing_arxiv_id":"2604.04037","citing_title":"Geometric Limits of Knowledge Distillation: A Minimum-Width Theorem via Superposition Theory","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01609","citing_title":"Concepts Whisper While Syntax Shouts: Spectral Anti-Concentration and the Dual Geometry of Transformer Representations","ref_index":50,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05223","citing_title":"Structural Instability of Feature Composition","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2605.02234","citing_title":"Bucketing the Good Apples: A Method for Diagnosing and Improving Causal Abstraction","ref_index":19,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/PRTWLIKRP2RBFQQEE6UKUWY6OE","json":"https://pith.science/pith/PRTWLIKRP2RBFQQEE6UKUWY6OE.json","graph_json":"https://pith.science/api/pith-number/PRTWLIKRP2RBFQQEE6UKUWY6OE/graph.json","events_json":"https://pith.science/api/pith-number/PRTWLIKRP2RBFQQEE6UKUWY6OE/events.json","paper":"https://pith.science/paper/PRTWLIKR"},"agent_actions":{"view_html":"https://pith.science/pith/PRTWLIKRP2RBFQQEE6UKUWY6OE","download_json":"https://pith.science/pith/PRTWLIKRP2RBFQQEE6UKUWY6OE.json","view_paper":"https://pith.science/paper/PRTWLIKR","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2210.01892&json=true","fetch_graph":"https://pith.science/api/pith-number/PRTWLIKRP2RBFQQEE6UKUWY6OE/graph.json","fetch_events":"https://pith.science/api/pith-number/PRTWLIKRP2RBFQQEE6UKUWY6OE/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/PRTWLIKRP2RBFQQEE6UKUWY6OE/action/timestamp_anchor","attest_storage":"https://pith.science/pith/PRTWLIKRP2RBFQQEE6UKUWY6OE/action/storage_attestation","attest_author":"https://pith.science/pith/PRTWLIKRP2RBFQQEE6UKUWY6OE/action/author_attestation","sign_citation":"https://pith.science/pith/PRTWLIKRP2RBFQQEE6UKUWY6OE/action/citation_signature","submit_replication":"https://pith.science/pith/PRTWLIKRP2RBFQQEE6UKUWY6OE/action/replication_record"}},"created_at":"2026-07-05T10:38:29.874800+00:00","updated_at":"2026-07-05T10:38:29.874800+00:00"}