{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:ZN6VHYAAOJWASGRVVZLCI2VSQL","short_pith_number":"pith:ZN6VHYAA","schema_version":"1.0","canonical_sha256":"cb7d53e000726c091a35ae56246ab282d717fae9271fd022fe1acb904c33fdcc","source":{"kind":"arxiv","id":"2411.02631","version":1},"attestation_state":"computed","paper":{"title":"Extracting Unlearned Information from LLMs with Activation Steering","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Aleksei Kuvshinov, Atakan Seyito\\u{g}lu, Leo Schwinn, Stephan G\\\"unnemann","submitted_at":"2024-11-04T21:42:56Z","abstract_excerpt":"An unintended consequence of the vast pretraining of Large Language Models (LLMs) is the verbatim memorization of fragments of their training data, which may contain sensitive or copyrighted information. In recent years, unlearning has emerged as a solution to effectively remove sensitive knowledge from models after training. Yet, recent work has shown that supposedly deleted information can still be extracted by malicious actors through various attacks. Still, current attacks retrieve sets of possible candidate generations and are unable to pinpoint the output that contains the actual target "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2411.02631","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-11-04T21:42:56Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"17d54c2290c20d4b056b8571923329a3ea7f35e4029cab294eb2327ce64e8422","abstract_canon_sha256":"4559a08ccdc7b4027788bf83c1b5ad7cec97e1f101b284cacc8c714561e43517"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:31:15.027099Z","signature_b64":"7FNWKJZdan7n+Rp03Sn+KlQuhqRiu9H7fS93iPH6hbJUyqWNlx0VSUVz7yswRpc0himUL2E/tk17Wp+IZNYSBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"cb7d53e000726c091a35ae56246ab282d717fae9271fd022fe1acb904c33fdcc","last_reissued_at":"2026-07-05T09:31:15.026625Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:31:15.026625Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Extracting Unlearned Information from LLMs with Activation Steering","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Aleksei Kuvshinov, Atakan Seyito\\u{g}lu, Leo Schwinn, Stephan G\\\"unnemann","submitted_at":"2024-11-04T21:42:56Z","abstract_excerpt":"An unintended consequence of the vast pretraining of Large Language Models (LLMs) is the verbatim memorization of fragments of their training data, which may contain sensitive or copyrighted information. In recent years, unlearning has emerged as a solution to effectively remove sensitive knowledge from models after training. Yet, recent work has shown that supposedly deleted information can still be extracted by malicious actors through various attacks. Still, current attacks retrieve sets of possible candidate generations and are unable to pinpoint the output that contains the actual target "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2411.02631","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2411.02631/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2411.02631","created_at":"2026-07-05T09:31:15.026675+00:00"},{"alias_kind":"arxiv_version","alias_value":"2411.02631v1","created_at":"2026-07-05T09:31:15.026675+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2411.02631","created_at":"2026-07-05T09:31:15.026675+00:00"},{"alias_kind":"pith_short_12","alias_value":"ZN6VHYAAOJWA","created_at":"2026-07-05T09:31:15.026675+00:00"},{"alias_kind":"pith_short_16","alias_value":"ZN6VHYAAOJWASGRV","created_at":"2026-07-05T09:31:15.026675+00:00"},{"alias_kind":"pith_short_8","alias_value":"ZN6VHYAA","created_at":"2026-07-05T09:31:15.026675+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.24614","citing_title":"Measuring the Depth of LLM Unlearning via Activation Patching","ref_index":30,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ZN6VHYAAOJWASGRVVZLCI2VSQL","json":"https://pith.science/pith/ZN6VHYAAOJWASGRVVZLCI2VSQL.json","graph_json":"https://pith.science/api/pith-number/ZN6VHYAAOJWASGRVVZLCI2VSQL/graph.json","events_json":"https://pith.science/api/pith-number/ZN6VHYAAOJWASGRVVZLCI2VSQL/events.json","paper":"https://pith.science/paper/ZN6VHYAA"},"agent_actions":{"view_html":"https://pith.science/pith/ZN6VHYAAOJWASGRVVZLCI2VSQL","download_json":"https://pith.science/pith/ZN6VHYAAOJWASGRVVZLCI2VSQL.json","view_paper":"https://pith.science/paper/ZN6VHYAA","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2411.02631&json=true","fetch_graph":"https://pith.science/api/pith-number/ZN6VHYAAOJWASGRVVZLCI2VSQL/graph.json","fetch_events":"https://pith.science/api/pith-number/ZN6VHYAAOJWASGRVVZLCI2VSQL/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ZN6VHYAAOJWASGRVVZLCI2VSQL/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ZN6VHYAAOJWASGRVVZLCI2VSQL/action/storage_attestation","attest_author":"https://pith.science/pith/ZN6VHYAAOJWASGRVVZLCI2VSQL/action/author_attestation","sign_citation":"https://pith.science/pith/ZN6VHYAAOJWASGRVVZLCI2VSQL/action/citation_signature","submit_replication":"https://pith.science/pith/ZN6VHYAAOJWASGRVVZLCI2VSQL/action/replication_record"}},"created_at":"2026-07-05T09:31:15.026675+00:00","updated_at":"2026-07-05T09:31:15.026675+00:00"}