{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:27NM2QFFJXYPCAEKERPN2JW3KW","short_pith_number":"pith:27NM2QFF","schema_version":"1.0","canonical_sha256":"d7dacd40a54df0f1008a245edd26db5587e44183b169e5b11ce96cad5bef3655","source":{"kind":"arxiv","id":"2403.19521","version":4},"attestation_state":"computed","paper":{"title":"Interpreting Key Mechanisms of Factual Recall in Transformer-Based Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Ang Lv, Jian Xie, Ji-Rong Wen, Kaiyi Zhang, Lifeng Liu, Rui Yan, Yuhan Chen, Yulong Wang","submitted_at":"2024-03-28T15:54:59Z","abstract_excerpt":"In this paper, we delve into several mechanisms employed by Transformer-based language models (LLMs) for factual recall tasks. We outline a pipeline consisting of three major steps: (1) Given a prompt ``The capital of France is,'' task-specific attention heads extract the topic token, such as ``France,'' from the context and pass it to subsequent MLPs. (2) As attention heads' outputs are aggregated with equal weight and added to the residual stream, the subsequent MLP acts as an ``activation,'' which either erases or amplifies the information originating from individual heads. As a result, the"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2403.19521","kind":"arxiv","version":4},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-03-28T15:54:59Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"4d3352c920aa459b9a25300c1b0f66676e8e63aa146e0b3ccf9757042be3b63e","abstract_canon_sha256":"e3b5962c6ef2cefdf1b67dba50c09317435391576191ec718a557f7a9155f38f"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:22:39.055438Z","signature_b64":"oGovJX1DVKQ6sZobuNI/cmOUIhfSC8iPXutPgq3o28hDo2vaKh+WWKRgj0TX7fsTsSE2CQ4nQRefF0hooeMuDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"d7dacd40a54df0f1008a245edd26db5587e44183b169e5b11ce96cad5bef3655","last_reissued_at":"2026-07-05T08:22:39.054852Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:22:39.054852Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Interpreting Key Mechanisms of Factual Recall in Transformer-Based Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Ang Lv, Jian Xie, Ji-Rong Wen, Kaiyi Zhang, Lifeng Liu, Rui Yan, Yuhan Chen, Yulong Wang","submitted_at":"2024-03-28T15:54:59Z","abstract_excerpt":"In this paper, we delve into several mechanisms employed by Transformer-based language models (LLMs) for factual recall tasks. We outline a pipeline consisting of three major steps: (1) Given a prompt ``The capital of France is,'' task-specific attention heads extract the topic token, such as ``France,'' from the context and pass it to subsequent MLPs. (2) As attention heads' outputs are aggregated with equal weight and added to the residual stream, the subsequent MLP acts as an ``activation,'' which either erases or amplifies the information originating from individual heads. As a result, the"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2403.19521","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2403.19521/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2403.19521","created_at":"2026-07-05T08:22:39.054921+00:00"},{"alias_kind":"arxiv_version","alias_value":"2403.19521v4","created_at":"2026-07-05T08:22:39.054921+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2403.19521","created_at":"2026-07-05T08:22:39.054921+00:00"},{"alias_kind":"pith_short_12","alias_value":"27NM2QFFJXYP","created_at":"2026-07-05T08:22:39.054921+00:00"},{"alias_kind":"pith_short_16","alias_value":"27NM2QFFJXYPCAEK","created_at":"2026-07-05T08:22:39.054921+00:00"},{"alias_kind":"pith_short_8","alias_value":"27NM2QFF","created_at":"2026-07-05T08:22:39.054921+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":9,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.07502","citing_title":"Your UnEmbedding Matrix is Secretly a Feature Lens for Text Embeddings","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2605.22532","citing_title":"Relational Linear Properties in Language Models: An Empirical Investigation","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2606.00570","citing_title":"Revisiting Parameter-Based Knowledge Editing in Large Language Models: Theoretical Limits and Empirical Evidence","ref_index":61,"is_internal_anchor":false},{"citing_arxiv_id":"2605.22532","citing_title":"Relational Linear Properties in Language Models: An Empirical Investigation","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2510.26745","citing_title":"Deep sequence models tend to memorize geometrically; it is unclear why","ref_index":110,"is_internal_anchor":false},{"citing_arxiv_id":"2601.14004","citing_title":"Locate, Steer, and Improve: A Practical Survey of Actionable Mechanistic Interpretability in Large Language Models","ref_index":201,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10970","citing_title":"Context-Gated Associative Retrieval: From Theory to Transformers","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2604.12426","citing_title":"Do Transformers Use their Depth Adaptively? Evidence from a Relational Reasoning Task","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2604.19934","citing_title":"Tracing Relational Knowledge Recall in Large Language Models","ref_index":15,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/27NM2QFFJXYPCAEKERPN2JW3KW","json":"https://pith.science/pith/27NM2QFFJXYPCAEKERPN2JW3KW.json","graph_json":"https://pith.science/api/pith-number/27NM2QFFJXYPCAEKERPN2JW3KW/graph.json","events_json":"https://pith.science/api/pith-number/27NM2QFFJXYPCAEKERPN2JW3KW/events.json","paper":"https://pith.science/paper/27NM2QFF"},"agent_actions":{"view_html":"https://pith.science/pith/27NM2QFFJXYPCAEKERPN2JW3KW","download_json":"https://pith.science/pith/27NM2QFFJXYPCAEKERPN2JW3KW.json","view_paper":"https://pith.science/paper/27NM2QFF","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2403.19521&json=true","fetch_graph":"https://pith.science/api/pith-number/27NM2QFFJXYPCAEKERPN2JW3KW/graph.json","fetch_events":"https://pith.science/api/pith-number/27NM2QFFJXYPCAEKERPN2JW3KW/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/27NM2QFFJXYPCAEKERPN2JW3KW/action/timestamp_anchor","attest_storage":"https://pith.science/pith/27NM2QFFJXYPCAEKERPN2JW3KW/action/storage_attestation","attest_author":"https://pith.science/pith/27NM2QFFJXYPCAEKERPN2JW3KW/action/author_attestation","sign_citation":"https://pith.science/pith/27NM2QFFJXYPCAEKERPN2JW3KW/action/citation_signature","submit_replication":"https://pith.science/pith/27NM2QFFJXYPCAEKERPN2JW3KW/action/replication_record"}},"created_at":"2026-07-05T08:22:39.054921+00:00","updated_at":"2026-07-05T08:22:39.054921+00:00"}