{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:AAS3TGS2IOZXPGMVOUELOMK53B","short_pith_number":"pith:AAS3TGS2","schema_version":"1.0","canonical_sha256":"0025b99a5a43b37799957508b7315dd87b8aa6bc63b359b586a03214f71099fe","source":{"kind":"arxiv","id":"2302.00539","version":4},"attestation_state":"computed","paper":{"title":"Analyzing Leakage of Personally Identifiable Information in Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Ahmed Salem, Lukas Wutschitz, Nils Lukas, Robert Sim, Santiago Zanella-B\\'eguelin, Shruti Tople","submitted_at":"2023-02-01T16:04:48Z","abstract_excerpt":"Language Models (LMs) have been shown to leak information about training data through sentence-level membership inference and reconstruction attacks. Understanding the risk of LMs leaking Personally Identifiable Information (PII) has received less attention, which can be attributed to the false assumption that dataset curation techniques such as scrubbing are sufficient to prevent PII leakage. Scrubbing techniques reduce but do not prevent the risk of PII leakage: in practice scrubbing is imperfect and must balance the trade-off between minimizing disclosure and preserving the utility of the d"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2302.00539","kind":"arxiv","version":4},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2023-02-01T16:04:48Z","cross_cats_sorted":[],"title_canon_sha256":"09894ee46a08000cce8facc61b77991b6997e3fc0f6b8fa06d22baa53c8ac203","abstract_canon_sha256":"2764a03930d705bee1313adaf18bc96a99c0d9159c2ccd8b77c5438b22d26bbb"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:03:27.516406Z","signature_b64":"gbXvGLS5xmWBcEOC2ODRANp9wBr2Dx/hX/0JiRkk1wCdBO17swKpyzBXwVdkcUPKscZvfukTyuh7vAOoEAhGDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"0025b99a5a43b37799957508b7315dd87b8aa6bc63b359b586a03214f71099fe","last_reissued_at":"2026-07-05T06:03:27.515868Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:03:27.515868Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Analyzing Leakage of Personally Identifiable Information in Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Ahmed Salem, Lukas Wutschitz, Nils Lukas, Robert Sim, Santiago Zanella-B\\'eguelin, Shruti Tople","submitted_at":"2023-02-01T16:04:48Z","abstract_excerpt":"Language Models (LMs) have been shown to leak information about training data through sentence-level membership inference and reconstruction attacks. Understanding the risk of LMs leaking Personally Identifiable Information (PII) has received less attention, which can be attributed to the false assumption that dataset curation techniques such as scrubbing are sufficient to prevent PII leakage. Scrubbing techniques reduce but do not prevent the risk of PII leakage: in practice scrubbing is imperfect and must balance the trade-off between minimizing disclosure and preserving the utility of the d"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2302.00539","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2302.00539/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2302.00539","created_at":"2026-07-05T06:03:27.515934+00:00"},{"alias_kind":"arxiv_version","alias_value":"2302.00539v4","created_at":"2026-07-05T06:03:27.515934+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2302.00539","created_at":"2026-07-05T06:03:27.515934+00:00"},{"alias_kind":"pith_short_12","alias_value":"AAS3TGS2IOZX","created_at":"2026-07-05T06:03:27.515934+00:00"},{"alias_kind":"pith_short_16","alias_value":"AAS3TGS2IOZXPGMV","created_at":"2026-07-05T06:03:27.515934+00:00"},{"alias_kind":"pith_short_8","alias_value":"AAS3TGS2","created_at":"2026-07-05T06:03:27.515934+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.26627","citing_title":"Agents That Know Too Much: A Data-Centric Survey of Privacy in LLM Agents","ref_index":77,"is_internal_anchor":false},{"citing_arxiv_id":"2606.09590","citing_title":"Clinically Grounded Privacy Evaluation of Medical LMs","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2606.07595","citing_title":"VisualLeakBench: Reproducible Action-Boundary Propagation Failures in Vision-Language Agents","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2307.02483","citing_title":"Jailbroken: How Does LLM Safety Training Fail?","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2604.20720","citing_title":"COMPASS: COntinual Multilingual PEFT with Adaptive Semantic Sampling","ref_index":43,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/AAS3TGS2IOZXPGMVOUELOMK53B","json":"https://pith.science/pith/AAS3TGS2IOZXPGMVOUELOMK53B.json","graph_json":"https://pith.science/api/pith-number/AAS3TGS2IOZXPGMVOUELOMK53B/graph.json","events_json":"https://pith.science/api/pith-number/AAS3TGS2IOZXPGMVOUELOMK53B/events.json","paper":"https://pith.science/paper/AAS3TGS2"},"agent_actions":{"view_html":"https://pith.science/pith/AAS3TGS2IOZXPGMVOUELOMK53B","download_json":"https://pith.science/pith/AAS3TGS2IOZXPGMVOUELOMK53B.json","view_paper":"https://pith.science/paper/AAS3TGS2","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2302.00539&json=true","fetch_graph":"https://pith.science/api/pith-number/AAS3TGS2IOZXPGMVOUELOMK53B/graph.json","fetch_events":"https://pith.science/api/pith-number/AAS3TGS2IOZXPGMVOUELOMK53B/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/AAS3TGS2IOZXPGMVOUELOMK53B/action/timestamp_anchor","attest_storage":"https://pith.science/pith/AAS3TGS2IOZXPGMVOUELOMK53B/action/storage_attestation","attest_author":"https://pith.science/pith/AAS3TGS2IOZXPGMVOUELOMK53B/action/author_attestation","sign_citation":"https://pith.science/pith/AAS3TGS2IOZXPGMVOUELOMK53B/action/citation_signature","submit_replication":"https://pith.science/pith/AAS3TGS2IOZXPGMVOUELOMK53B/action/replication_record"}},"created_at":"2026-07-05T06:03:27.515934+00:00","updated_at":"2026-07-05T06:03:27.515934+00:00"}