{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:7MN3OT4N6NOJM525PWZGNQHXB4","short_pith_number":"pith:7MN3OT4N","schema_version":"1.0","canonical_sha256":"fb1bb74f8df35c96775d7db266c0f70f2feb972caadc437a2a36400df960a7c1","source":{"kind":"arxiv","id":"2310.01424","version":2},"attestation_state":"computed","paper":{"title":"Identifying and Mitigating Privacy Risks Stemming from Language Models: A Survey","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Adrian Weller, Ali Shahin Shamsabadi, Carolyn Ashurst, Victoria Smith","submitted_at":"2023-09-27T15:15:23Z","abstract_excerpt":"Large Language Models (LLMs) have shown greatly enhanced performance in recent years, attributed to increased size and extensive training data. This advancement has led to widespread interest and adoption across industries and the public. However, training data memorization in Machine Learning models scales with model size, particularly concerning for LLMs. Memorized text sequences have the potential to be directly leaked from LLMs, posing a serious threat to data privacy. Various techniques have been developed to attack LLMs and extract their training data. As these models continue to grow, t"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2310.01424","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2023-09-27T15:15:23Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"f31d1d68a36277bfec84f9e68cb2e4e817cb7ef46331c4768d47948ac9ace202","abstract_canon_sha256":"923bed0575dd3f79d25cedcd9da086fdf2622c4e24d8da3618d8304844f329e6"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:33:32.397088Z","signature_b64":"TC0/uMZ1jSs71ky9SUqb9mR0fOYzYt8s3QlTgvJZ5aVc/IxaVSEaeF7+LJjm5AYptY4mqr5ccrqLvgL5HG92CA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"fb1bb74f8df35c96775d7db266c0f70f2feb972caadc437a2a36400df960a7c1","last_reissued_at":"2026-07-05T08:33:32.396561Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:33:32.396561Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Identifying and Mitigating Privacy Risks Stemming from Language Models: A Survey","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Adrian Weller, Ali Shahin Shamsabadi, Carolyn Ashurst, Victoria Smith","submitted_at":"2023-09-27T15:15:23Z","abstract_excerpt":"Large Language Models (LLMs) have shown greatly enhanced performance in recent years, attributed to increased size and extensive training data. This advancement has led to widespread interest and adoption across industries and the public. However, training data memorization in Machine Learning models scales with model size, particularly concerning for LLMs. Memorized text sequences have the potential to be directly leaked from LLMs, posing a serious threat to data privacy. Various techniques have been developed to attack LLMs and extract their training data. As these models continue to grow, t"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2310.01424","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2310.01424/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2310.01424","created_at":"2026-07-05T08:33:32.396625+00:00"},{"alias_kind":"arxiv_version","alias_value":"2310.01424v2","created_at":"2026-07-05T08:33:32.396625+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2310.01424","created_at":"2026-07-05T08:33:32.396625+00:00"},{"alias_kind":"pith_short_12","alias_value":"7MN3OT4N6NOJ","created_at":"2026-07-05T08:33:32.396625+00:00"},{"alias_kind":"pith_short_16","alias_value":"7MN3OT4N6NOJM525","created_at":"2026-07-05T08:33:32.396625+00:00"},{"alias_kind":"pith_short_8","alias_value":"7MN3OT4N","created_at":"2026-07-05T08:33:32.396625+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2512.06713","citing_title":"Look Twice before You Leap: A Rational Framework for Localized Adversarial Anonymization","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2512.05929","citing_title":"LLM Harms: A Taxonomy and Discussion","ref_index":101,"is_internal_anchor":false},{"citing_arxiv_id":"2404.13501","citing_title":"A Survey on the Memory Mechanism of Large Language Model based Agents","ref_index":67,"is_internal_anchor":false},{"citing_arxiv_id":"2604.09695","citing_title":"Assessing Privacy Preservation and Utility in Online Vision-Language Models","ref_index":6,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/7MN3OT4N6NOJM525PWZGNQHXB4","json":"https://pith.science/pith/7MN3OT4N6NOJM525PWZGNQHXB4.json","graph_json":"https://pith.science/api/pith-number/7MN3OT4N6NOJM525PWZGNQHXB4/graph.json","events_json":"https://pith.science/api/pith-number/7MN3OT4N6NOJM525PWZGNQHXB4/events.json","paper":"https://pith.science/paper/7MN3OT4N"},"agent_actions":{"view_html":"https://pith.science/pith/7MN3OT4N6NOJM525PWZGNQHXB4","download_json":"https://pith.science/pith/7MN3OT4N6NOJM525PWZGNQHXB4.json","view_paper":"https://pith.science/paper/7MN3OT4N","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2310.01424&json=true","fetch_graph":"https://pith.science/api/pith-number/7MN3OT4N6NOJM525PWZGNQHXB4/graph.json","fetch_events":"https://pith.science/api/pith-number/7MN3OT4N6NOJM525PWZGNQHXB4/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/7MN3OT4N6NOJM525PWZGNQHXB4/action/timestamp_anchor","attest_storage":"https://pith.science/pith/7MN3OT4N6NOJM525PWZGNQHXB4/action/storage_attestation","attest_author":"https://pith.science/pith/7MN3OT4N6NOJM525PWZGNQHXB4/action/author_attestation","sign_citation":"https://pith.science/pith/7MN3OT4N6NOJM525PWZGNQHXB4/action/citation_signature","submit_replication":"https://pith.science/pith/7MN3OT4N6NOJM525PWZGNQHXB4/action/replication_record"}},"created_at":"2026-07-05T08:33:32.396625+00:00","updated_at":"2026-07-05T08:33:32.396625+00:00"}