{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:XNQCZQMDCDBPV2CTMRSL4GE7NZ","short_pith_number":"pith:XNQCZQMD","schema_version":"1.0","canonical_sha256":"bb602cc18310c2fae8536464be189f6e43ed91cb7ecc8a1976c7afd2d61fe4f5","source":{"kind":"arxiv","id":"2503.09532","version":4},"attestation_state":"computed","paper":{"title":"SAEBench: A Comprehensive Benchmark for Sparse Autoencoders in Language Model Interpretability","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.LG","authors_text":"Adam Karvonen, Arthur Conmy, Callum McDougall, Can Rager, Curt Tigges, David Chanin, Demian Till, Eoin Farrell, Johnny Lin, Joseph Bloom, Kola Ayonrinde, Matthew Wearden, Neel Nanda, Samuel Marks, Yeu-Tong Lau","submitted_at":"2025-03-12T16:49:02Z","abstract_excerpt":"Sparse autoencoders (SAEs) are a popular technique for interpreting language model activations, and there is extensive recent work on improving SAE effectiveness. However, most prior work evaluates progress using unsupervised proxy metrics with unclear practical relevance. We introduce SAEBench, a comprehensive evaluation suite that measures SAE performance across eight diverse metrics, spanning interpretability, feature disentanglement and practical applications like unlearning. To enable systematic comparison, we open-source a suite of over 200 SAEs across eight recently proposed SAE archite"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2503.09532","kind":"arxiv","version":4},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-03-12T16:49:02Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"a5153deb32be16d9be3df98efbae33fbf1211e24bffa369f3bf9fbeb89f6f17d","abstract_canon_sha256":"41f83990de24d01ad0789c9238c54ba95022606c106eddb0522c754ebbe4789e"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:15:44.944013Z","signature_b64":"39Z3YoMqzba03jsnV2peKJ4Rzsn7RMiEb4TZSihIm/5I8erbgz4GAZDGt/g7Y3/koWyXYkM2VIXUHK15o5AiBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"bb602cc18310c2fae8536464be189f6e43ed91cb7ecc8a1976c7afd2d61fe4f5","last_reissued_at":"2026-07-05T11:15:44.943500Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:15:44.943500Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"SAEBench: A Comprehensive Benchmark for Sparse Autoencoders in Language Model Interpretability","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.LG","authors_text":"Adam Karvonen, Arthur Conmy, Callum McDougall, Can Rager, Curt Tigges, David Chanin, Demian Till, Eoin Farrell, Johnny Lin, Joseph Bloom, Kola Ayonrinde, Matthew Wearden, Neel Nanda, Samuel Marks, Yeu-Tong Lau","submitted_at":"2025-03-12T16:49:02Z","abstract_excerpt":"Sparse autoencoders (SAEs) are a popular technique for interpreting language model activations, and there is extensive recent work on improving SAE effectiveness. However, most prior work evaluates progress using unsupervised proxy metrics with unclear practical relevance. We introduce SAEBench, a comprehensive evaluation suite that measures SAE performance across eight diverse metrics, spanning interpretability, feature disentanglement and practical applications like unlearning. To enable systematic comparison, we open-source a suite of over 200 SAEs across eight recently proposed SAE archite"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2503.09532","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2503.09532/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2503.09532","created_at":"2026-07-05T11:15:44.943568+00:00"},{"alias_kind":"arxiv_version","alias_value":"2503.09532v4","created_at":"2026-07-05T11:15:44.943568+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2503.09532","created_at":"2026-07-05T11:15:44.943568+00:00"},{"alias_kind":"pith_short_12","alias_value":"XNQCZQMDCDBP","created_at":"2026-07-05T11:15:44.943568+00:00"},{"alias_kind":"pith_short_16","alias_value":"XNQCZQMDCDBPV2CT","created_at":"2026-07-05T11:15:44.943568+00:00"},{"alias_kind":"pith_short_8","alias_value":"XNQCZQMD","created_at":"2026-07-05T11:15:44.943568+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":11,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.22994","citing_title":"Do Sparse Autoencoders Learn Meaningful Concept Hierarchies?","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2606.09653","citing_title":"A Unifying Framework for Concept-Based Representational Similarity","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2605.28149","citing_title":"Sign-Aware Gated Sparse Autoencoders: Modeling Anticorrelated Features with Bi-Jump-ReLU Activations","ref_index":41,"is_internal_anchor":false},{"citing_arxiv_id":"2509.18127","citing_title":"Safe-SAIL: Towards a Fine-grained Safety Landscape of Large Language Models via Sparse Autoencoder Interpretation Framework","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2601.14004","citing_title":"Locate, Steer, and Improve: A Practical Survey of Actionable Mechanistic Interpretability in Large Language Models","ref_index":148,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07922","citing_title":"Tree SAE: Learning Hierarchical Feature Structures in Sparse Autoencoders","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10536","citing_title":"HH-SAE: Discovering and Steering Hierarchical Knowledge of Complex Manifolds","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06494","citing_title":"From Token Lists to Graph Motifs: Weisfeiler-Lehman Analysis of Sparse Autoencoder Features","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2604.08846","citing_title":"Dictionary-Aligned Concept Control for Safeguarding Multimodal LLMs","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07922","citing_title":"Tree SAE: Learning Hierarchical Feature Structures in Sparse Autoencoders","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05223","citing_title":"Structural Instability of Feature Composition","ref_index":4,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/XNQCZQMDCDBPV2CTMRSL4GE7NZ","json":"https://pith.science/pith/XNQCZQMDCDBPV2CTMRSL4GE7NZ.json","graph_json":"https://pith.science/api/pith-number/XNQCZQMDCDBPV2CTMRSL4GE7NZ/graph.json","events_json":"https://pith.science/api/pith-number/XNQCZQMDCDBPV2CTMRSL4GE7NZ/events.json","paper":"https://pith.science/paper/XNQCZQMD"},"agent_actions":{"view_html":"https://pith.science/pith/XNQCZQMDCDBPV2CTMRSL4GE7NZ","download_json":"https://pith.science/pith/XNQCZQMDCDBPV2CTMRSL4GE7NZ.json","view_paper":"https://pith.science/paper/XNQCZQMD","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2503.09532&json=true","fetch_graph":"https://pith.science/api/pith-number/XNQCZQMDCDBPV2CTMRSL4GE7NZ/graph.json","fetch_events":"https://pith.science/api/pith-number/XNQCZQMDCDBPV2CTMRSL4GE7NZ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/XNQCZQMDCDBPV2CTMRSL4GE7NZ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/XNQCZQMDCDBPV2CTMRSL4GE7NZ/action/storage_attestation","attest_author":"https://pith.science/pith/XNQCZQMDCDBPV2CTMRSL4GE7NZ/action/author_attestation","sign_citation":"https://pith.science/pith/XNQCZQMDCDBPV2CTMRSL4GE7NZ/action/citation_signature","submit_replication":"https://pith.science/pith/XNQCZQMDCDBPV2CTMRSL4GE7NZ/action/replication_record"}},"created_at":"2026-07-05T11:15:44.943568+00:00","updated_at":"2026-07-05T11:15:44.943568+00:00"}