{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:LTYGAJASFQKQNQ6TG2P3ZXX37U","short_pith_number":"pith:LTYGAJAS","schema_version":"1.0","canonical_sha256":"5cf06024122c1506c3d3369fbcdefbfd116fb686ffdc7188e61b5741fcb9e194","source":{"kind":"arxiv","id":"2506.23845","version":1},"attestation_state":"computed","paper":{"title":"Use Sparse Autoencoders to Discover Unknown Concepts, Not to Act on Known Concepts","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.CY"],"primary_cat":"cs.LG","authors_text":"Emma Pierson, Jon Kleinberg, Kenny Peng, Nikhil Garg, Rajiv Movva","submitted_at":"2025-06-30T13:35:56Z","abstract_excerpt":"While sparse autoencoders (SAEs) have generated significant excitement, a series of negative results have added to skepticism about their usefulness. Here, we establish a conceptual distinction that reconciles competing narratives surrounding SAEs. We argue that while SAEs may be less effective for acting on known concepts, SAEs are powerful tools for discovering unknown concepts. This distinction cleanly separates existing negative and positive results, and suggests several classes of SAE applications. Specifically, we outline use cases for SAEs in (i) ML interpretability, explainability, fai"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2506.23845","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-06-30T13:35:56Z","cross_cats_sorted":["cs.AI","cs.CL","cs.CY"],"title_canon_sha256":"3008860709b0975531ce54ed810d55cdbb430a69e0b2876a7f355e4501181e9e","abstract_canon_sha256":"e97747b82e53132afae25fc9011d729d22372ba6d1725f41af49edaaf39d4e99"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:29:28.855337Z","signature_b64":"F4lWKoWDTzaxT+IbD2UQfiCHyWAZQLDDWUSZRQA/Qss9NILcf8zYXbsXoFJ5CgvI9UkZbAc7SsR8/k+ds10bBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"5cf06024122c1506c3d3369fbcdefbfd116fb686ffdc7188e61b5741fcb9e194","last_reissued_at":"2026-07-05T11:29:28.854794Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:29:28.854794Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Use Sparse Autoencoders to Discover Unknown Concepts, Not to Act on Known Concepts","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.CY"],"primary_cat":"cs.LG","authors_text":"Emma Pierson, Jon Kleinberg, Kenny Peng, Nikhil Garg, Rajiv Movva","submitted_at":"2025-06-30T13:35:56Z","abstract_excerpt":"While sparse autoencoders (SAEs) have generated significant excitement, a series of negative results have added to skepticism about their usefulness. Here, we establish a conceptual distinction that reconciles competing narratives surrounding SAEs. We argue that while SAEs may be less effective for acting on known concepts, SAEs are powerful tools for discovering unknown concepts. This distinction cleanly separates existing negative and positive results, and suggests several classes of SAE applications. Specifically, we outline use cases for SAEs in (i) ML interpretability, explainability, fai"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.23845","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2506.23845/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2506.23845","created_at":"2026-07-05T11:29:28.854868+00:00"},{"alias_kind":"arxiv_version","alias_value":"2506.23845v1","created_at":"2026-07-05T11:29:28.854868+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.23845","created_at":"2026-07-05T11:29:28.854868+00:00"},{"alias_kind":"pith_short_12","alias_value":"LTYGAJASFQKQ","created_at":"2026-07-05T11:29:28.854868+00:00"},{"alias_kind":"pith_short_16","alias_value":"LTYGAJASFQKQNQ6T","created_at":"2026-07-05T11:29:28.854868+00:00"},{"alias_kind":"pith_short_8","alias_value":"LTYGAJAS","created_at":"2026-07-05T11:29:28.854868+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":6,"sample":[{"citing_arxiv_id":"2606.05750","citing_title":"Three Years of r/ChatGPT: Societal Impact Evaluations from Social Media Data","ref_index":12,"is_internal_anchor":true},{"citing_arxiv_id":"2606.03029","citing_title":"Conditional Hypothesis Generation for LLM-Based Text Analysis with Researcher-Specified Covariates","ref_index":2,"is_internal_anchor":true},{"citing_arxiv_id":"2605.12991","citing_title":"Not Just RLHF: Why Alignment Alone Won't Fix Multi-Agent Sycophancy","ref_index":27,"is_internal_anchor":true},{"citing_arxiv_id":"2510.18184","citing_title":"ActivationReasoning: Logical Reasoning in Latent Activation Spaces","ref_index":13,"is_internal_anchor":true},{"citing_arxiv_id":"2511.01680","citing_title":"Making Interpretable Discoveries from Unstructured Data: A High-Dimensional Multiple Hypothesis Testing Approach","ref_index":37,"is_internal_anchor":true},{"citing_arxiv_id":"2605.12991","citing_title":"Not Just RLHF: Why Alignment Alone Won't Fix Multi-Agent Sycophancy","ref_index":27,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/LTYGAJASFQKQNQ6TG2P3ZXX37U","json":"https://pith.science/pith/LTYGAJASFQKQNQ6TG2P3ZXX37U.json","graph_json":"https://pith.science/api/pith-number/LTYGAJASFQKQNQ6TG2P3ZXX37U/graph.json","events_json":"https://pith.science/api/pith-number/LTYGAJASFQKQNQ6TG2P3ZXX37U/events.json","paper":"https://pith.science/paper/LTYGAJAS"},"agent_actions":{"view_html":"https://pith.science/pith/LTYGAJASFQKQNQ6TG2P3ZXX37U","download_json":"https://pith.science/pith/LTYGAJASFQKQNQ6TG2P3ZXX37U.json","view_paper":"https://pith.science/paper/LTYGAJAS","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2506.23845&json=true","fetch_graph":"https://pith.science/api/pith-number/LTYGAJASFQKQNQ6TG2P3ZXX37U/graph.json","fetch_events":"https://pith.science/api/pith-number/LTYGAJASFQKQNQ6TG2P3ZXX37U/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/LTYGAJASFQKQNQ6TG2P3ZXX37U/action/timestamp_anchor","attest_storage":"https://pith.science/pith/LTYGAJASFQKQNQ6TG2P3ZXX37U/action/storage_attestation","attest_author":"https://pith.science/pith/LTYGAJASFQKQNQ6TG2P3ZXX37U/action/author_attestation","sign_citation":"https://pith.science/pith/LTYGAJASFQKQNQ6TG2P3ZXX37U/action/citation_signature","submit_replication":"https://pith.science/pith/LTYGAJASFQKQNQ6TG2P3ZXX37U/action/replication_record"}},"created_at":"2026-07-05T11:29:28.854868+00:00","updated_at":"2026-07-05T11:29:28.854868+00:00"}