{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2020:PFHIOHXL2WCVGY3SX2THWSGSVY","short_pith_number":"pith:PFHIOHXL","schema_version":"1.0","canonical_sha256":"794e871eebd585536372bea67b48d2ae19c0e45d548b7600a379ab7b15c66a65","source":{"kind":"arxiv","id":"2012.00363","version":1},"attestation_state":"computed","paper":{"title":"Modifying Memories in Transformer Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Ankit Singh Rawat, Chen Zhu, Daliang Li, Felix Yu, Manzil Zaheer, Sanjiv Kumar, Srinadh Bhojanapalli","submitted_at":"2020-12-01T09:39:13Z","abstract_excerpt":"Large Transformer models have achieved impressive performance in many natural language tasks. In particular, Transformer based language models have been shown to have great capabilities in encoding factual knowledge in their vast amount of parameters. While the tasks of improving the memorization and generalization of Transformers have been widely studied, it is not well known how to make transformers forget specific old facts and memorize new ones. In this paper, we propose a new task of \\emph{explicitly modifying specific factual knowledge in Transformer models while ensuring the model perfo"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2012.00363","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2020-12-01T09:39:13Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"b85731976c2dae887d086072a1fb8ee0329febd7f3a5feb559b943988b8fcfbe","abstract_canon_sha256":"eca3de1a5d4fbff56226ba2a7dba937440f935285bb64078ee992c141121a656"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T01:55:35.519960Z","signature_b64":"ezSIm4ik1NOf6PdjGSyFSJ4SksyaIiT9UH3xb3+3V0GzGAi4Ybg95kdSOhXlcfScFl19TUAMarelXf/Y+S26Cg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"794e871eebd585536372bea67b48d2ae19c0e45d548b7600a379ab7b15c66a65","last_reissued_at":"2026-07-05T01:55:35.519468Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T01:55:35.519468Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Modifying Memories in Transformer Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Ankit Singh Rawat, Chen Zhu, Daliang Li, Felix Yu, Manzil Zaheer, Sanjiv Kumar, Srinadh Bhojanapalli","submitted_at":"2020-12-01T09:39:13Z","abstract_excerpt":"Large Transformer models have achieved impressive performance in many natural language tasks. In particular, Transformer based language models have been shown to have great capabilities in encoding factual knowledge in their vast amount of parameters. While the tasks of improving the memorization and generalization of Transformers have been widely studied, it is not well known how to make transformers forget specific old facts and memorize new ones. In this paper, we propose a new task of \\emph{explicitly modifying specific factual knowledge in Transformer models while ensuring the model perfo"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2012.00363","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2012.00363/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2012.00363","created_at":"2026-07-05T01:55:35.519528+00:00"},{"alias_kind":"arxiv_version","alias_value":"2012.00363v1","created_at":"2026-07-05T01:55:35.519528+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2012.00363","created_at":"2026-07-05T01:55:35.519528+00:00"},{"alias_kind":"pith_short_12","alias_value":"PFHIOHXL2WCV","created_at":"2026-07-05T01:55:35.519528+00:00"},{"alias_kind":"pith_short_16","alias_value":"PFHIOHXL2WCVGY3S","created_at":"2026-07-05T01:55:35.519528+00:00"},{"alias_kind":"pith_short_8","alias_value":"PFHIOHXL","created_at":"2026-07-05T01:55:35.519528+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":17,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.23276","citing_title":"Exposing the Illusion of Erasure in Knowledge Editing for LLMs","ref_index":48,"is_internal_anchor":false},{"citing_arxiv_id":"2606.22627","citing_title":"Orthogonal Representation Editing: Decoupling Semantic Entanglement in Batch Knowledge Editing of LLMs","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2606.14668","citing_title":"When to Write and When to Suppress: Route-Specialized Dual Adapters for Memory-Assisted Knowledge Editing","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2607.02052","citing_title":"Mitigating Package Hallucinations in Large Language Models via Model Editing","ref_index":50,"is_internal_anchor":false},{"citing_arxiv_id":"2606.03924","citing_title":"Knowledge Editing in Masked Diffusion Language Models","ref_index":66,"is_internal_anchor":false},{"citing_arxiv_id":"2606.02995","citing_title":"Patcher: Post-Hoc Patching of Backdoored Large Language Models","ref_index":57,"is_internal_anchor":false},{"citing_arxiv_id":"2606.03179","citing_title":"HyperPatch: Sequential Knowledge Editing Under n-ary Structural Drift","ref_index":43,"is_internal_anchor":false},{"citing_arxiv_id":"2606.14668","citing_title":"When to Write and When to Suppress: Route-Specialized Dual Adapters for Memory-Assisted Knowledge Editing","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2605.29826","citing_title":"Towards Localized and Disentangled Knowledge Editing for Multimodal Large Language Models","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2510.26745","citing_title":"Deep sequence models tend to memorize geometrically; it is unclear why","ref_index":208,"is_internal_anchor":false},{"citing_arxiv_id":"2602.01167","citing_title":"Do All Individual Layers Help? An Empirical Study of Task-Interfering Layers in Vision-Language Models","ref_index":59,"is_internal_anchor":false},{"citing_arxiv_id":"2605.16686","citing_title":"Scalable Knowledge Editing for Mixture-of-Experts LLMs via Tensor-Structured Updates","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2511.03182","citing_title":"Understanding Robustness of Model Editing in Code LLMs","ref_index":47,"is_internal_anchor":false},{"citing_arxiv_id":"2202.05262","citing_title":"Locating and Editing Factual Associations in GPT","ref_index":43,"is_internal_anchor":false},{"citing_arxiv_id":"2602.02543","citing_title":"Norm Anchors Make Model Edits Last","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2212.04089","citing_title":"Editing Models with Task Arithmetic","ref_index":111,"is_internal_anchor":false},{"citing_arxiv_id":"2604.11214","citing_title":"HiEdit: Lifelong Model Editing with Hierarchical Reinforcement Learning","ref_index":6,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/PFHIOHXL2WCVGY3SX2THWSGSVY","json":"https://pith.science/pith/PFHIOHXL2WCVGY3SX2THWSGSVY.json","graph_json":"https://pith.science/api/pith-number/PFHIOHXL2WCVGY3SX2THWSGSVY/graph.json","events_json":"https://pith.science/api/pith-number/PFHIOHXL2WCVGY3SX2THWSGSVY/events.json","paper":"https://pith.science/paper/PFHIOHXL"},"agent_actions":{"view_html":"https://pith.science/pith/PFHIOHXL2WCVGY3SX2THWSGSVY","download_json":"https://pith.science/pith/PFHIOHXL2WCVGY3SX2THWSGSVY.json","view_paper":"https://pith.science/paper/PFHIOHXL","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2012.00363&json=true","fetch_graph":"https://pith.science/api/pith-number/PFHIOHXL2WCVGY3SX2THWSGSVY/graph.json","fetch_events":"https://pith.science/api/pith-number/PFHIOHXL2WCVGY3SX2THWSGSVY/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/PFHIOHXL2WCVGY3SX2THWSGSVY/action/timestamp_anchor","attest_storage":"https://pith.science/pith/PFHIOHXL2WCVGY3SX2THWSGSVY/action/storage_attestation","attest_author":"https://pith.science/pith/PFHIOHXL2WCVGY3SX2THWSGSVY/action/author_attestation","sign_citation":"https://pith.science/pith/PFHIOHXL2WCVGY3SX2THWSGSVY/action/citation_signature","submit_replication":"https://pith.science/pith/PFHIOHXL2WCVGY3SX2THWSGSVY/action/replication_record"}},"created_at":"2026-07-05T01:55:35.519528+00:00","updated_at":"2026-07-05T01:55:35.519528+00:00"}