{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:PY2IEIF5LYL3PNKL6ZNWWCAMOJ","short_pith_number":"pith:PY2IEIF5","schema_version":"1.0","canonical_sha256":"7e348220bd5e17b7b54bf65b6b080c7254bbe5c36e56b41a0186c95c51677285","source":{"kind":"arxiv","id":"2503.11232","version":1},"attestation_state":"computed","paper":{"title":"PrivacyScalpel: Enhancing LLM Privacy via Interpretable Feature Intervention with Sparse Autoencoders","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.LG","authors_text":"Ahmed Frikha, Krishna Kanth Nakka, Muhammad Reza Ar Razi, Ricardo Mendes, Xuebing Zhou, Xue Jiang","submitted_at":"2025-03-14T09:31:01Z","abstract_excerpt":"Large Language Models (LLMs) have demonstrated remarkable capabilities in natural language processing but also pose significant privacy risks by memorizing and leaking Personally Identifiable Information (PII). Existing mitigation strategies, such as differential privacy and neuron-level interventions, often degrade model utility or fail to effectively prevent leakage. To address this challenge, we introduce PrivacyScalpel, a novel privacy-preserving framework that leverages LLM interpretability techniques to identify and mitigate PII leakage while maintaining performance. PrivacyScalpel compr"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2503.11232","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2025-03-14T09:31:01Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"f7beaef1d19b4e9bfefd683b04c8e8e11365a8bcf68f6d44f42b5ebc5572f318","abstract_canon_sha256":"96383dbfe6965092b35493e65848594e6d79c50e08ee695c79caaa7dc859ba97"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:31:14.885859Z","signature_b64":"6YB+JEq3YyH6M2rZ4zKdEP4DHQsPCL0Voy0gZPaVvx2Nw8Qkr6BZEjuK2UwG0zzEWg6tCtSuboPnessYUseCCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"7e348220bd5e17b7b54bf65b6b080c7254bbe5c36e56b41a0186c95c51677285","last_reissued_at":"2026-07-05T10:31:14.885381Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:31:14.885381Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"PrivacyScalpel: Enhancing LLM Privacy via Interpretable Feature Intervention with Sparse Autoencoders","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.LG","authors_text":"Ahmed Frikha, Krishna Kanth Nakka, Muhammad Reza Ar Razi, Ricardo Mendes, Xuebing Zhou, Xue Jiang","submitted_at":"2025-03-14T09:31:01Z","abstract_excerpt":"Large Language Models (LLMs) have demonstrated remarkable capabilities in natural language processing but also pose significant privacy risks by memorizing and leaking Personally Identifiable Information (PII). Existing mitigation strategies, such as differential privacy and neuron-level interventions, often degrade model utility or fail to effectively prevent leakage. To address this challenge, we introduce PrivacyScalpel, a novel privacy-preserving framework that leverages LLM interpretability techniques to identify and mitigate PII leakage while maintaining performance. PrivacyScalpel compr"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2503.11232","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2503.11232/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2503.11232","created_at":"2026-07-05T10:31:14.885435+00:00"},{"alias_kind":"arxiv_version","alias_value":"2503.11232v1","created_at":"2026-07-05T10:31:14.885435+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2503.11232","created_at":"2026-07-05T10:31:14.885435+00:00"},{"alias_kind":"pith_short_12","alias_value":"PY2IEIF5LYL3","created_at":"2026-07-05T10:31:14.885435+00:00"},{"alias_kind":"pith_short_16","alias_value":"PY2IEIF5LYL3PNKL","created_at":"2026-07-05T10:31:14.885435+00:00"},{"alias_kind":"pith_short_8","alias_value":"PY2IEIF5","created_at":"2026-07-05T10:31:14.885435+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2512.06655","citing_title":"Graph-Regularized Sparse Autoencoders for LLM Safety Steering","ref_index":6,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/PY2IEIF5LYL3PNKL6ZNWWCAMOJ","json":"https://pith.science/pith/PY2IEIF5LYL3PNKL6ZNWWCAMOJ.json","graph_json":"https://pith.science/api/pith-number/PY2IEIF5LYL3PNKL6ZNWWCAMOJ/graph.json","events_json":"https://pith.science/api/pith-number/PY2IEIF5LYL3PNKL6ZNWWCAMOJ/events.json","paper":"https://pith.science/paper/PY2IEIF5"},"agent_actions":{"view_html":"https://pith.science/pith/PY2IEIF5LYL3PNKL6ZNWWCAMOJ","download_json":"https://pith.science/pith/PY2IEIF5LYL3PNKL6ZNWWCAMOJ.json","view_paper":"https://pith.science/paper/PY2IEIF5","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2503.11232&json=true","fetch_graph":"https://pith.science/api/pith-number/PY2IEIF5LYL3PNKL6ZNWWCAMOJ/graph.json","fetch_events":"https://pith.science/api/pith-number/PY2IEIF5LYL3PNKL6ZNWWCAMOJ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/PY2IEIF5LYL3PNKL6ZNWWCAMOJ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/PY2IEIF5LYL3PNKL6ZNWWCAMOJ/action/storage_attestation","attest_author":"https://pith.science/pith/PY2IEIF5LYL3PNKL6ZNWWCAMOJ/action/author_attestation","sign_citation":"https://pith.science/pith/PY2IEIF5LYL3PNKL6ZNWWCAMOJ/action/citation_signature","submit_replication":"https://pith.science/pith/PY2IEIF5LYL3PNKL6ZNWWCAMOJ/action/replication_record"}},"created_at":"2026-07-05T10:31:14.885435+00:00","updated_at":"2026-07-05T10:31:14.885435+00:00"}