{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:G2VPENNR66SJ5UDHLFMITRB7UK","short_pith_number":"pith:G2VPENNR","schema_version":"1.0","canonical_sha256":"36aaf235b1f7a49ed067595889c43fa286fdbebdc3d8e9e5f5d124d488e6dbfa","source":{"kind":"arxiv","id":"2104.07143","version":1},"attestation_state":"computed","paper":{"title":"An Interpretability Illusion for BERT","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Adam Pearce, Andy Coenen, Ann Yuan, Emily Reif, Fernanda Vi\\'egas, Martin Wattenberg, Tolga Bolukbasi","submitted_at":"2021-04-14T22:04:48Z","abstract_excerpt":"We describe an \"interpretability illusion\" that arises when analyzing the BERT model. Activations of individual neurons in the network may spuriously appear to encode a single, simple concept, when in fact they are encoding something far more complex. The same effect holds for linear combinations of activations. We trace the source of this illusion to geometric properties of BERT's embedding space as well as the fact that common text corpora represent only narrow slices of possible English sentences. We provide a taxonomy of model-learned concepts and discuss methodological implications for in"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2104.07143","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2021-04-14T22:04:48Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"bb15ef7e0248510662a11010118aed5559dcbb9f5aa5dc67ec1c6fe8e191771f","abstract_canon_sha256":"d7b7b9b8986584d8209b25f71275087e51daf8185ec8a5c8ecbdadf0be0a2d6d"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T02:32:18.249765Z","signature_b64":"r3gJKMfpe9J2BAfgw27hFl5h0R7KlKMXtZQniDPtO2JoJgxUDgB154hyUFLCXvyXaa/vUEJaWyKewP/tnMQCCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"36aaf235b1f7a49ed067595889c43fa286fdbebdc3d8e9e5f5d124d488e6dbfa","last_reissued_at":"2026-07-05T02:32:18.249339Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T02:32:18.249339Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"An Interpretability Illusion for BERT","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Adam Pearce, Andy Coenen, Ann Yuan, Emily Reif, Fernanda Vi\\'egas, Martin Wattenberg, Tolga Bolukbasi","submitted_at":"2021-04-14T22:04:48Z","abstract_excerpt":"We describe an \"interpretability illusion\" that arises when analyzing the BERT model. Activations of individual neurons in the network may spuriously appear to encode a single, simple concept, when in fact they are encoding something far more complex. The same effect holds for linear combinations of activations. We trace the source of this illusion to geometric properties of BERT's embedding space as well as the fact that common text corpora represent only narrow slices of possible English sentences. We provide a taxonomy of model-learned concepts and discuss methodological implications for in"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2104.07143","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2104.07143/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2104.07143","created_at":"2026-07-05T02:32:18.249403+00:00"},{"alias_kind":"arxiv_version","alias_value":"2104.07143v1","created_at":"2026-07-05T02:32:18.249403+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2104.07143","created_at":"2026-07-05T02:32:18.249403+00:00"},{"alias_kind":"pith_short_12","alias_value":"G2VPENNR66SJ","created_at":"2026-07-05T02:32:18.249403+00:00"},{"alias_kind":"pith_short_16","alias_value":"G2VPENNR66SJ5UDH","created_at":"2026-07-05T02:32:18.249403+00:00"},{"alias_kind":"pith_short_8","alias_value":"G2VPENNR","created_at":"2026-07-05T02:32:18.249403+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.06267","citing_title":"Many Circuits, One Mechanism: Input Variation and Evaluation Granularity in Circuit Discovery","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2605.28149","citing_title":"Sign-Aware Gated Sparse Autoencoders: Modeling Anticorrelated Features with Bi-Jump-ReLU Activations","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2404.16014","citing_title":"Improving Dictionary Learning with Gated Sparse Autoencoders","ref_index":202,"is_internal_anchor":false},{"citing_arxiv_id":"2304.05969","citing_title":"Localizing Model Behavior with Path Patching","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2211.00593","citing_title":"Interpretability in the Wild: a Circuit for Indirect Object Identification in GPT-2 small","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2406.04093","citing_title":"Scaling and evaluating sparse autoencoders","ref_index":6,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/G2VPENNR66SJ5UDHLFMITRB7UK","json":"https://pith.science/pith/G2VPENNR66SJ5UDHLFMITRB7UK.json","graph_json":"https://pith.science/api/pith-number/G2VPENNR66SJ5UDHLFMITRB7UK/graph.json","events_json":"https://pith.science/api/pith-number/G2VPENNR66SJ5UDHLFMITRB7UK/events.json","paper":"https://pith.science/paper/G2VPENNR"},"agent_actions":{"view_html":"https://pith.science/pith/G2VPENNR66SJ5UDHLFMITRB7UK","download_json":"https://pith.science/pith/G2VPENNR66SJ5UDHLFMITRB7UK.json","view_paper":"https://pith.science/paper/G2VPENNR","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2104.07143&json=true","fetch_graph":"https://pith.science/api/pith-number/G2VPENNR66SJ5UDHLFMITRB7UK/graph.json","fetch_events":"https://pith.science/api/pith-number/G2VPENNR66SJ5UDHLFMITRB7UK/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/G2VPENNR66SJ5UDHLFMITRB7UK/action/timestamp_anchor","attest_storage":"https://pith.science/pith/G2VPENNR66SJ5UDHLFMITRB7UK/action/storage_attestation","attest_author":"https://pith.science/pith/G2VPENNR66SJ5UDHLFMITRB7UK/action/author_attestation","sign_citation":"https://pith.science/pith/G2VPENNR66SJ5UDHLFMITRB7UK/action/citation_signature","submit_replication":"https://pith.science/pith/G2VPENNR66SJ5UDHLFMITRB7UK/action/replication_record"}},"created_at":"2026-07-05T02:32:18.249403+00:00","updated_at":"2026-07-05T02:32:18.249403+00:00"}