{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:NFN3OZ7AAC4F7LBZJYJS56SIYO","short_pith_number":"pith:NFN3OZ7A","schema_version":"1.0","canonical_sha256":"695bb767e000b85fac394e132efa48c397585f609e3a5be779ad5f424ad57c45","source":{"kind":"arxiv","id":"2310.07298","version":2},"attestation_state":"computed","paper":{"title":"Beyond Memorization: Violating Privacy Via Inference with Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.AI","authors_text":"Mark Vero, Martin Vechev, Mislav Balunovi\\'c, Robin Staab","submitted_at":"2023-10-11T08:32:46Z","abstract_excerpt":"Current privacy research on large language models (LLMs) primarily focuses on the issue of extracting memorized training data. At the same time, models' inference capabilities have increased drastically. This raises the key question of whether current LLMs could violate individuals' privacy by inferring personal attributes from text given at inference time. In this work, we present the first comprehensive study on the capabilities of pretrained LLMs to infer personal attributes from text. We construct a dataset consisting of real Reddit profiles, and show that current LLMs can infer a wide ran"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2310.07298","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2023-10-11T08:32:46Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"fecbe9ed7928ac24b6ba8297a13618553481457bdc1997096ee3cf741c86b8bd","abstract_canon_sha256":"c03aeb1f1b8048668b00b48a563b9a040c11e64b8623fc0ffeb2c57981a38560"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:15:40.290032Z","signature_b64":"EDNxFrLQflMk6qqB/PGsI0Ey6qgsj/eNE7q/G13t1ojV5rLcxtrnyLO4xTCsmmoXEorjwBYrimOd9edT5pKNDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"695bb767e000b85fac394e132efa48c397585f609e3a5be779ad5f424ad57c45","last_reissued_at":"2026-07-05T08:15:40.289366Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:15:40.289366Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Beyond Memorization: Violating Privacy Via Inference with Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.AI","authors_text":"Mark Vero, Martin Vechev, Mislav Balunovi\\'c, Robin Staab","submitted_at":"2023-10-11T08:32:46Z","abstract_excerpt":"Current privacy research on large language models (LLMs) primarily focuses on the issue of extracting memorized training data. At the same time, models' inference capabilities have increased drastically. This raises the key question of whether current LLMs could violate individuals' privacy by inferring personal attributes from text given at inference time. In this work, we present the first comprehensive study on the capabilities of pretrained LLMs to infer personal attributes from text. We construct a dataset consisting of real Reddit profiles, and show that current LLMs can infer a wide ran"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2310.07298","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2310.07298/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2310.07298","created_at":"2026-07-05T08:15:40.289512+00:00"},{"alias_kind":"arxiv_version","alias_value":"2310.07298v2","created_at":"2026-07-05T08:15:40.289512+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2310.07298","created_at":"2026-07-05T08:15:40.289512+00:00"},{"alias_kind":"pith_short_12","alias_value":"NFN3OZ7AAC4F","created_at":"2026-07-05T08:15:40.289512+00:00"},{"alias_kind":"pith_short_16","alias_value":"NFN3OZ7AAC4F7LBZ","created_at":"2026-07-05T08:15:40.289512+00:00"},{"alias_kind":"pith_short_8","alias_value":"NFN3OZ7A","created_at":"2026-07-05T08:15:40.289512+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":15,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.12341","citing_title":"OCELOT: Inference-Leakage Budgets for Privacy-Preserving LLM Agents","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2606.06755","citing_title":"PromptPrint: Behavioral Biometrics Through Natural Language Prompting in LLMs","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2606.03399","citing_title":"Selective Token-Level Cryptographic Redaction for Privacy-Preserving Clinical Deployment of Large Language Models","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2606.27936","citing_title":"Agentic AI-Powered Re-Identification: An Emerging, Scalable Threat to Mobility Microdata Privacy","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2606.00540","citing_title":"Trustworthy Recommendation in the Era of Large Language Models: Opportunities and Challenges","ref_index":213,"is_internal_anchor":false},{"citing_arxiv_id":"2605.23640","citing_title":"CachePrune: Privacy-Aware and Fine-Grained KV Cache Sharing for Efficient LLM Inference","ref_index":52,"is_internal_anchor":false},{"citing_arxiv_id":"2605.23820","citing_title":"Inferential Privacy Leakage in Anonymized Conversational AI Logs","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2510.00761","citing_title":"Downgrade to Upgrade: Optimizer Simplification Enhances Robustness in LLM Unlearning","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2604.27438","citing_title":"Tracking Conversations: Measuring Content and Identity Exposure on AI Chatbots","ref_index":66,"is_internal_anchor":false},{"citing_arxiv_id":"2604.27438","citing_title":"Tracking Conversations: Measuring Content and Identity Exposure on AI Chatbots","ref_index":66,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10013","citing_title":"When Are LLM Inferences Acceptable? User Reactions and Control Preferences for Inferred Personal Information","ref_index":55,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06232","citing_title":"Profiling for Pennies: Unveiling the Privacy Iceberg of LLM Agents","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2604.11523","citing_title":"PAC-BENCH: Evaluating Multi-Agent Collaboration under Privacy Constraints","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2604.09952","citing_title":"SLM Finetuning for Natural Language to Domain Specific Code Generation in Production","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2604.05571","citing_title":"Understanding User Privacy Perceptions of GenAI Smartphones","ref_index":71,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/NFN3OZ7AAC4F7LBZJYJS56SIYO","json":"https://pith.science/pith/NFN3OZ7AAC4F7LBZJYJS56SIYO.json","graph_json":"https://pith.science/api/pith-number/NFN3OZ7AAC4F7LBZJYJS56SIYO/graph.json","events_json":"https://pith.science/api/pith-number/NFN3OZ7AAC4F7LBZJYJS56SIYO/events.json","paper":"https://pith.science/paper/NFN3OZ7A"},"agent_actions":{"view_html":"https://pith.science/pith/NFN3OZ7AAC4F7LBZJYJS56SIYO","download_json":"https://pith.science/pith/NFN3OZ7AAC4F7LBZJYJS56SIYO.json","view_paper":"https://pith.science/paper/NFN3OZ7A","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2310.07298&json=true","fetch_graph":"https://pith.science/api/pith-number/NFN3OZ7AAC4F7LBZJYJS56SIYO/graph.json","fetch_events":"https://pith.science/api/pith-number/NFN3OZ7AAC4F7LBZJYJS56SIYO/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/NFN3OZ7AAC4F7LBZJYJS56SIYO/action/timestamp_anchor","attest_storage":"https://pith.science/pith/NFN3OZ7AAC4F7LBZJYJS56SIYO/action/storage_attestation","attest_author":"https://pith.science/pith/NFN3OZ7AAC4F7LBZJYJS56SIYO/action/author_attestation","sign_citation":"https://pith.science/pith/NFN3OZ7AAC4F7LBZJYJS56SIYO/action/citation_signature","submit_replication":"https://pith.science/pith/NFN3OZ7AAC4F7LBZJYJS56SIYO/action/replication_record"}},"created_at":"2026-07-05T08:15:40.289512+00:00","updated_at":"2026-07-05T08:15:40.289512+00:00"}