{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:Y4BPGKMVT36E22ZCZY64TBFR3J","short_pith_number":"pith:Y4BPGKMV","schema_version":"1.0","canonical_sha256":"c702f329959efc4d6b22ce3dc984b1da7bd4c6d2f2d60cd956da1bf325932ca2","source":{"kind":"arxiv","id":"2210.03588","version":3},"attestation_state":"computed","paper":{"title":"Understanding Transformer Memorization Recall Through Idioms","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Adi Haviv, Ido Cohen, Jacob Gidron, Mor Geva, Roei Schuster, Yoav Goldberg","submitted_at":"2022-10-07T14:45:31Z","abstract_excerpt":"To produce accurate predictions, language models (LMs) must balance between generalization and memorization. Yet, little is known about the mechanism by which transformer LMs employ their memorization capacity. When does a model decide to output a memorized phrase, and how is this phrase then retrieved from memory? In this work, we offer the first methodological framework for probing and characterizing recall of memorized sequences in transformer LMs. First, we lay out criteria for detecting model inputs that trigger memory recall, and propose idioms as inputs that typically fulfill these crit"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2210.03588","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2022-10-07T14:45:31Z","cross_cats_sorted":[],"title_canon_sha256":"c5ee5150b16b6326a521216281fb9bc2a293c6c259a81e524574fc4cb3abebf1","abstract_canon_sha256":"95a78f9fd507b66780472574e04d9e2a2e009d28bb6d55c14c1de7d53beb0f5e"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:40:49.911905Z","signature_b64":"YN9tn+scdGLeBh1Zlv10fJ6XEQJ5RiEv2t7UaPz1vThwPuDIoHGFdTdZ9edLe8EEvgNvM0+w6pB9AMeztHfVDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"c702f329959efc4d6b22ce3dc984b1da7bd4c6d2f2d60cd956da1bf325932ca2","last_reissued_at":"2026-07-05T05:40:49.911405Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:40:49.911405Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Understanding Transformer Memorization Recall Through Idioms","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Adi Haviv, Ido Cohen, Jacob Gidron, Mor Geva, Roei Schuster, Yoav Goldberg","submitted_at":"2022-10-07T14:45:31Z","abstract_excerpt":"To produce accurate predictions, language models (LMs) must balance between generalization and memorization. Yet, little is known about the mechanism by which transformer LMs employ their memorization capacity. When does a model decide to output a memorized phrase, and how is this phrase then retrieved from memory? In this work, we offer the first methodological framework for probing and characterizing recall of memorized sequences in transformer LMs. First, we lay out criteria for detecting model inputs that trigger memory recall, and propose idioms as inputs that typically fulfill these crit"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2210.03588","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2210.03588/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2210.03588","created_at":"2026-07-05T05:40:49.911463+00:00"},{"alias_kind":"arxiv_version","alias_value":"2210.03588v3","created_at":"2026-07-05T05:40:49.911463+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2210.03588","created_at":"2026-07-05T05:40:49.911463+00:00"},{"alias_kind":"pith_short_12","alias_value":"Y4BPGKMVT36E","created_at":"2026-07-05T05:40:49.911463+00:00"},{"alias_kind":"pith_short_16","alias_value":"Y4BPGKMVT36E22ZC","created_at":"2026-07-05T05:40:49.911463+00:00"},{"alias_kind":"pith_short_8","alias_value":"Y4BPGKMV","created_at":"2026-07-05T05:40:49.911463+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2304.01373","citing_title":"Pythia: A Suite for Analyzing Large Language Models Across Training and Scaling","ref_index":181,"is_internal_anchor":false},{"citing_arxiv_id":"2310.11511","citing_title":"Self-RAG: Learning to Retrieve, Generate, and Critique through Self-Reflection","ref_index":68,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/Y4BPGKMVT36E22ZCZY64TBFR3J","json":"https://pith.science/pith/Y4BPGKMVT36E22ZCZY64TBFR3J.json","graph_json":"https://pith.science/api/pith-number/Y4BPGKMVT36E22ZCZY64TBFR3J/graph.json","events_json":"https://pith.science/api/pith-number/Y4BPGKMVT36E22ZCZY64TBFR3J/events.json","paper":"https://pith.science/paper/Y4BPGKMV"},"agent_actions":{"view_html":"https://pith.science/pith/Y4BPGKMVT36E22ZCZY64TBFR3J","download_json":"https://pith.science/pith/Y4BPGKMVT36E22ZCZY64TBFR3J.json","view_paper":"https://pith.science/paper/Y4BPGKMV","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2210.03588&json=true","fetch_graph":"https://pith.science/api/pith-number/Y4BPGKMVT36E22ZCZY64TBFR3J/graph.json","fetch_events":"https://pith.science/api/pith-number/Y4BPGKMVT36E22ZCZY64TBFR3J/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/Y4BPGKMVT36E22ZCZY64TBFR3J/action/timestamp_anchor","attest_storage":"https://pith.science/pith/Y4BPGKMVT36E22ZCZY64TBFR3J/action/storage_attestation","attest_author":"https://pith.science/pith/Y4BPGKMVT36E22ZCZY64TBFR3J/action/author_attestation","sign_citation":"https://pith.science/pith/Y4BPGKMVT36E22ZCZY64TBFR3J/action/citation_signature","submit_replication":"https://pith.science/pith/Y4BPGKMVT36E22ZCZY64TBFR3J/action/replication_record"}},"created_at":"2026-07-05T05:40:49.911463+00:00","updated_at":"2026-07-05T05:40:49.911463+00:00"}