{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:IPF5UAANGNGCCITPD6GW52RBC5","short_pith_number":"pith:IPF5UAAN","schema_version":"1.0","canonical_sha256":"43cbda000d334c21226f1f8d6eea21177fe8fffde6c1414b2ea8a70cd4c9ff17","source":{"kind":"arxiv","id":"2104.08696","version":2},"attestation_state":"computed","paper":{"title":"Knowledge Neurons in Pretrained Transformers","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Baobao Chang, Damai Dai, Furu Wei, Li Dong, Yaru Hao, Zhifang Sui","submitted_at":"2021-04-18T03:38:26Z","abstract_excerpt":"Large-scale pretrained language models are surprisingly good at recalling factual knowledge presented in the training corpus. In this paper, we present preliminary studies on how factual knowledge is stored in pretrained Transformers by introducing the concept of knowledge neurons. Specifically, we examine the fill-in-the-blank cloze task for BERT. Given a relational fact, we propose a knowledge attribution method to identify the neurons that express the fact. We find that the activation of such knowledge neurons is positively correlated to the expression of their corresponding facts. In our c"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2104.08696","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2021-04-18T03:38:26Z","cross_cats_sorted":[],"title_canon_sha256":"9f3cad8ae4b8cd9c3c44ac7c115258d8824c2fe82a81e92ff353948bcb0db5b5","abstract_canon_sha256":"70cac5060bbb9f1d761f841acf0eeca44ffce62c5e5f46b07468b3408ff6bdf7"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T04:03:24.728749Z","signature_b64":"eBN7mxHLkDPtnQTDk3lasxISWUsWCnihRE27anO2KXxO7TgVmB2Zled/QnCZdm3GRNIlyht2/sBJnOQVr6XIBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"43cbda000d334c21226f1f8d6eea21177fe8fffde6c1414b2ea8a70cd4c9ff17","last_reissued_at":"2026-07-05T04:03:24.728255Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T04:03:24.728255Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Knowledge Neurons in Pretrained Transformers","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Baobao Chang, Damai Dai, Furu Wei, Li Dong, Yaru Hao, Zhifang Sui","submitted_at":"2021-04-18T03:38:26Z","abstract_excerpt":"Large-scale pretrained language models are surprisingly good at recalling factual knowledge presented in the training corpus. In this paper, we present preliminary studies on how factual knowledge is stored in pretrained Transformers by introducing the concept of knowledge neurons. Specifically, we examine the fill-in-the-blank cloze task for BERT. Given a relational fact, we propose a knowledge attribution method to identify the neurons that express the fact. We find that the activation of such knowledge neurons is positively correlated to the expression of their corresponding facts. In our c"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2104.08696","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2104.08696/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2104.08696","created_at":"2026-07-05T04:03:24.728306+00:00"},{"alias_kind":"arxiv_version","alias_value":"2104.08696v2","created_at":"2026-07-05T04:03:24.728306+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2104.08696","created_at":"2026-07-05T04:03:24.728306+00:00"},{"alias_kind":"pith_short_12","alias_value":"IPF5UAANGNGC","created_at":"2026-07-05T04:03:24.728306+00:00"},{"alias_kind":"pith_short_16","alias_value":"IPF5UAANGNGCCITP","created_at":"2026-07-05T04:03:24.728306+00:00"},{"alias_kind":"pith_short_8","alias_value":"IPF5UAAN","created_at":"2026-07-05T04:03:24.728306+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":13,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.23276","citing_title":"Exposing the Illusion of Erasure in Knowledge Editing for LLMs","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2606.21345","citing_title":"Factual Retrieval in LLMs Is a Redundant, Distributed and Non-Contiguous Process","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2606.04662","citing_title":"Why Muon Outperforms Adam: A Curvature Perspective","ref_index":52,"is_internal_anchor":false},{"citing_arxiv_id":"2606.02090","citing_title":"FocusDiT: Masking Queries in Diffusion Transformers for Fine-grained Image Generation","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2606.31099","citing_title":"Seeing Through Multiple Views: Parameter-Efficient Fine-Tuning via Selective Neurons for Consistent Radiology Report Generation","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2204.06745","citing_title":"GPT-NeoX-20B: An Open-Source Autoregressive Language Model","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2506.13727","citing_title":"Attribution-Guided Pruning for Insight and Control: Circuit Discovery and Targeted Correction in Small-scale LLMs","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2310.12508","citing_title":"SalUn: Empowering Machine Unlearning via Gradient-based Weight Saliency in Both Image Classification and Generation","ref_index":138,"is_internal_anchor":false},{"citing_arxiv_id":"2304.01373","citing_title":"Pythia: A Suite for Analyzing Large Language Models Across Training and Scaling","ref_index":146,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12299","citing_title":"GKnow: Measuring the Entanglement of Gender Bias and Factual Gender","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2604.23750","citing_title":"The Override Gap: A Magnitude Account of Knowledge Conflict Failure in Hypernetwork-Based Instant LLM Adaptation","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2604.23750","citing_title":"The Override Gap: A Magnitude Account of Knowledge Conflict Failure in Hypernetwork-Based Instant LLM Adaptation","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2604.17396","citing_title":"Representation-Guided Parameter-Efficient LLM Unlearning","ref_index":104,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/IPF5UAANGNGCCITPD6GW52RBC5","json":"https://pith.science/pith/IPF5UAANGNGCCITPD6GW52RBC5.json","graph_json":"https://pith.science/api/pith-number/IPF5UAANGNGCCITPD6GW52RBC5/graph.json","events_json":"https://pith.science/api/pith-number/IPF5UAANGNGCCITPD6GW52RBC5/events.json","paper":"https://pith.science/paper/IPF5UAAN"},"agent_actions":{"view_html":"https://pith.science/pith/IPF5UAANGNGCCITPD6GW52RBC5","download_json":"https://pith.science/pith/IPF5UAANGNGCCITPD6GW52RBC5.json","view_paper":"https://pith.science/paper/IPF5UAAN","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2104.08696&json=true","fetch_graph":"https://pith.science/api/pith-number/IPF5UAANGNGCCITPD6GW52RBC5/graph.json","fetch_events":"https://pith.science/api/pith-number/IPF5UAANGNGCCITPD6GW52RBC5/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/IPF5UAANGNGCCITPD6GW52RBC5/action/timestamp_anchor","attest_storage":"https://pith.science/pith/IPF5UAANGNGCCITPD6GW52RBC5/action/storage_attestation","attest_author":"https://pith.science/pith/IPF5UAANGNGCCITPD6GW52RBC5/action/author_attestation","sign_citation":"https://pith.science/pith/IPF5UAANGNGCCITPD6GW52RBC5/action/citation_signature","submit_replication":"https://pith.science/pith/IPF5UAANGNGCCITPD6GW52RBC5/action/replication_record"}},"created_at":"2026-07-05T04:03:24.728306+00:00","updated_at":"2026-07-05T04:03:24.728306+00:00"}