{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:UD2ZU6EUTFIO4SYCPGD5U2BJ72","short_pith_number":"pith:UD2ZU6EU","schema_version":"1.0","canonical_sha256":"a0f59a78949950ee4b027987da6829fe9f29857a6383aef43ce9d8819e6cd951","source":{"kind":"arxiv","id":"2112.12938","version":2},"attestation_state":"computed","paper":{"title":"Counterfactual Memorization in Neural Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Chiyuan Zhang, Daphne Ippolito, Florian Tram\\`er, Katherine Lee, Matthew Jagielski, Nicholas Carlini","submitted_at":"2021-12-24T04:20:57Z","abstract_excerpt":"Modern neural language models that are widely used in various NLP tasks risk memorizing sensitive information from their training data. Understanding this memorization is important in real world applications and also from a learning-theoretical perspective. An open question in previous studies of language model memorization is how to filter out \"common\" memorization. In fact, most memorization criteria strongly correlate with the number of occurrences in the training set, capturing memorized familiar phrases, public knowledge, templated texts, or other repeated data. We formulate a notion of c"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2112.12938","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2021-12-24T04:20:57Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"3af9378afd284ac30016c758f15162375cfe02634f3add7ec36f15584921907e","abstract_canon_sha256":"e86a9a07a73915a9c7ac47eeb2313f56bc2cb011004e024dd3a234e11e46236b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:00:42.589968Z","signature_b64":"5s6pnBeoCkfIrrB3snn1SlnJpeelq3lUita2JkVfc9ASd1CMAKAtRuotP0AbtlEHRV46hU+rddMgZ01LtPcnBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"a0f59a78949950ee4b027987da6829fe9f29857a6383aef43ce9d8819e6cd951","last_reissued_at":"2026-07-05T07:00:42.589480Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:00:42.589480Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Counterfactual Memorization in Neural Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Chiyuan Zhang, Daphne Ippolito, Florian Tram\\`er, Katherine Lee, Matthew Jagielski, Nicholas Carlini","submitted_at":"2021-12-24T04:20:57Z","abstract_excerpt":"Modern neural language models that are widely used in various NLP tasks risk memorizing sensitive information from their training data. Understanding this memorization is important in real world applications and also from a learning-theoretical perspective. An open question in previous studies of language model memorization is how to filter out \"common\" memorization. In fact, most memorization criteria strongly correlate with the number of occurrences in the training set, capturing memorized familiar phrases, public knowledge, templated texts, or other repeated data. We formulate a notion of c"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2112.12938","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2112.12938/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2112.12938","created_at":"2026-07-05T07:00:42.589534+00:00"},{"alias_kind":"arxiv_version","alias_value":"2112.12938v2","created_at":"2026-07-05T07:00:42.589534+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2112.12938","created_at":"2026-07-05T07:00:42.589534+00:00"},{"alias_kind":"pith_short_12","alias_value":"UD2ZU6EUTFIO","created_at":"2026-07-05T07:00:42.589534+00:00"},{"alias_kind":"pith_short_16","alias_value":"UD2ZU6EUTFIO4SYC","created_at":"2026-07-05T07:00:42.589534+00:00"},{"alias_kind":"pith_short_8","alias_value":"UD2ZU6EU","created_at":"2026-07-05T07:00:42.589534+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":7,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2204.06745","citing_title":"GPT-NeoX-20B: An Open-Source Autoregressive Language Model","ref_index":108,"is_internal_anchor":false},{"citing_arxiv_id":"2501.02407","citing_title":"Towards the Anonymization of the Language Modeling","ref_index":67,"is_internal_anchor":false},{"citing_arxiv_id":"2305.15717","citing_title":"The False Promise of Imitating Proprietary LLMs","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2308.05374","citing_title":"Trustworthy LLMs: a Survey and Guideline for Evaluating Large Language Models' Alignment","ref_index":185,"is_internal_anchor":false},{"citing_arxiv_id":"2310.16789","citing_title":"Detecting Pretraining Data from Large Language Models","ref_index":119,"is_internal_anchor":false},{"citing_arxiv_id":"2202.07646","citing_title":"Quantifying Memorization Across Neural Language Models","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2310.11511","citing_title":"Self-RAG: Learning to Retrieve, Generate, and Critique through Self-Reflection","ref_index":65,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/UD2ZU6EUTFIO4SYCPGD5U2BJ72","json":"https://pith.science/pith/UD2ZU6EUTFIO4SYCPGD5U2BJ72.json","graph_json":"https://pith.science/api/pith-number/UD2ZU6EUTFIO4SYCPGD5U2BJ72/graph.json","events_json":"https://pith.science/api/pith-number/UD2ZU6EUTFIO4SYCPGD5U2BJ72/events.json","paper":"https://pith.science/paper/UD2ZU6EU"},"agent_actions":{"view_html":"https://pith.science/pith/UD2ZU6EUTFIO4SYCPGD5U2BJ72","download_json":"https://pith.science/pith/UD2ZU6EUTFIO4SYCPGD5U2BJ72.json","view_paper":"https://pith.science/paper/UD2ZU6EU","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2112.12938&json=true","fetch_graph":"https://pith.science/api/pith-number/UD2ZU6EUTFIO4SYCPGD5U2BJ72/graph.json","fetch_events":"https://pith.science/api/pith-number/UD2ZU6EUTFIO4SYCPGD5U2BJ72/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/UD2ZU6EUTFIO4SYCPGD5U2BJ72/action/timestamp_anchor","attest_storage":"https://pith.science/pith/UD2ZU6EUTFIO4SYCPGD5U2BJ72/action/storage_attestation","attest_author":"https://pith.science/pith/UD2ZU6EUTFIO4SYCPGD5U2BJ72/action/author_attestation","sign_citation":"https://pith.science/pith/UD2ZU6EUTFIO4SYCPGD5U2BJ72/action/citation_signature","submit_replication":"https://pith.science/pith/UD2ZU6EUTFIO4SYCPGD5U2BJ72/action/replication_record"}},"created_at":"2026-07-05T07:00:42.589534+00:00","updated_at":"2026-07-05T07:00:42.589534+00:00"}