{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:NQ5C7WQUFLUVDCSJRKORDI7OZP","short_pith_number":"pith:NQ5C7WQU","schema_version":"1.0","canonical_sha256":"6c3a2fda142ae9518a498a9d11a3eecbe385320f630f9b713017a67748f30b22","source":{"kind":"arxiv","id":"2504.08192","version":1},"attestation_state":"computed","paper":{"title":"SAEs $\\textit{Can}$ Improve Unlearning: Dynamic Sparse Autoencoder Guardrails for Precision Unlearning in LLMs","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.CR"],"primary_cat":"cs.LG","authors_text":"Aashiq Muhamed, Jacopo Bonato, Mona Diab, Virginia Smith","submitted_at":"2025-04-11T01:24:03Z","abstract_excerpt":"Machine unlearning is a promising approach to improve LLM safety by removing unwanted knowledge from the model. However, prevailing gradient-based unlearning methods suffer from issues such as high computational costs, hyperparameter instability, poor sequential unlearning capability, vulnerability to relearning attacks, low data efficiency, and lack of interpretability. While Sparse Autoencoders are well-suited to improve these aspects by enabling targeted activation-based unlearning, prior approaches underperform gradient-based methods. This work demonstrates that, contrary to these earlier "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2504.08192","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-04-11T01:24:03Z","cross_cats_sorted":["cs.AI","cs.CL","cs.CR"],"title_canon_sha256":"6a1201e5bbe184605e2b9b5cfa28a40e812faf566a8d2250b292397dfda296bf","abstract_canon_sha256":"4595fd4a3e249c02ad7018052f95803610b52a52790ad8f9e5cfe490c1c46b80"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:47:44.734583Z","signature_b64":"mJb9GtrcSPqASaNxPL2eZD21C1cxYm3qpB/y9Fe9zai4+mLGKKyECPOomsasritAs9usfdtWaY+NtLy6QZCnCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"6c3a2fda142ae9518a498a9d11a3eecbe385320f630f9b713017a67748f30b22","last_reissued_at":"2026-07-05T10:47:44.734124Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:47:44.734124Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"SAEs $\\textit{Can}$ Improve Unlearning: Dynamic Sparse Autoencoder Guardrails for Precision Unlearning in LLMs","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.CR"],"primary_cat":"cs.LG","authors_text":"Aashiq Muhamed, Jacopo Bonato, Mona Diab, Virginia Smith","submitted_at":"2025-04-11T01:24:03Z","abstract_excerpt":"Machine unlearning is a promising approach to improve LLM safety by removing unwanted knowledge from the model. However, prevailing gradient-based unlearning methods suffer from issues such as high computational costs, hyperparameter instability, poor sequential unlearning capability, vulnerability to relearning attacks, low data efficiency, and lack of interpretability. While Sparse Autoencoders are well-suited to improve these aspects by enabling targeted activation-based unlearning, prior approaches underperform gradient-based methods. This work demonstrates that, contrary to these earlier "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2504.08192","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2504.08192/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2504.08192","created_at":"2026-07-05T10:47:44.734181+00:00"},{"alias_kind":"arxiv_version","alias_value":"2504.08192v1","created_at":"2026-07-05T10:47:44.734181+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2504.08192","created_at":"2026-07-05T10:47:44.734181+00:00"},{"alias_kind":"pith_short_12","alias_value":"NQ5C7WQUFLUV","created_at":"2026-07-05T10:47:44.734181+00:00"},{"alias_kind":"pith_short_16","alias_value":"NQ5C7WQUFLUVDCSJ","created_at":"2026-07-05T10:47:44.734181+00:00"},{"alias_kind":"pith_short_8","alias_value":"NQ5C7WQU","created_at":"2026-07-05T10:47:44.734181+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2601.14004","citing_title":"Locate, Steer, and Improve: A Practical Survey of Actionable Mechanistic Interpretability in Large Language Models","ref_index":216,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/NQ5C7WQUFLUVDCSJRKORDI7OZP","json":"https://pith.science/pith/NQ5C7WQUFLUVDCSJRKORDI7OZP.json","graph_json":"https://pith.science/api/pith-number/NQ5C7WQUFLUVDCSJRKORDI7OZP/graph.json","events_json":"https://pith.science/api/pith-number/NQ5C7WQUFLUVDCSJRKORDI7OZP/events.json","paper":"https://pith.science/paper/NQ5C7WQU"},"agent_actions":{"view_html":"https://pith.science/pith/NQ5C7WQUFLUVDCSJRKORDI7OZP","download_json":"https://pith.science/pith/NQ5C7WQUFLUVDCSJRKORDI7OZP.json","view_paper":"https://pith.science/paper/NQ5C7WQU","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2504.08192&json=true","fetch_graph":"https://pith.science/api/pith-number/NQ5C7WQUFLUVDCSJRKORDI7OZP/graph.json","fetch_events":"https://pith.science/api/pith-number/NQ5C7WQUFLUVDCSJRKORDI7OZP/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/NQ5C7WQUFLUVDCSJRKORDI7OZP/action/timestamp_anchor","attest_storage":"https://pith.science/pith/NQ5C7WQUFLUVDCSJRKORDI7OZP/action/storage_attestation","attest_author":"https://pith.science/pith/NQ5C7WQUFLUVDCSJRKORDI7OZP/action/author_attestation","sign_citation":"https://pith.science/pith/NQ5C7WQUFLUVDCSJRKORDI7OZP/action/citation_signature","submit_replication":"https://pith.science/pith/NQ5C7WQUFLUVDCSJRKORDI7OZP/action/replication_record"}},"created_at":"2026-07-05T10:47:44.734181+00:00","updated_at":"2026-07-05T10:47:44.734181+00:00"}