{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:3MYT5KLZSGZHH4NKYWIVALH2L3","short_pith_number":"pith:3MYT5KLZ","schema_version":"1.0","canonical_sha256":"db313ea97991b273f1aac591502cfa5eef2159f0dbc64c213f7855da2362f275","source":{"kind":"arxiv","id":"2410.13928","version":3},"attestation_state":"computed","paper":{"title":"Automatically Interpreting Millions of Features in Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.LG","authors_text":"Alex Mallen, Caden Juang, Gon\\c{c}alo Paulo, Nora Belrose","submitted_at":"2024-10-17T17:56:01Z","abstract_excerpt":"While the activations of neurons in deep neural networks usually do not have a simple human-understandable interpretation, sparse autoencoders (SAEs) can be used to transform these activations into a higher-dimensional latent space which may be more easily interpretable. However, these SAEs can have millions of distinct latent features, making it infeasible for humans to manually interpret each one. In this work, we build an open-source automated pipeline to generate and evaluate natural language explanations for SAE features using LLMs. We test our framework on SAEs of varying sizes, activati"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.13928","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-10-17T17:56:01Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"2624aae6b6c6554bb839ded4c822b2c866c91ee7602169ba2b31e2ed33681181","abstract_canon_sha256":"09976c11e711bdcac16da1422a9dcbc518a5d204462b1876e0fab8e7f6bc5eb3"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:49:31.868294Z","signature_b64":"ZYxfw35aJdWphr8uKhY5OD2sJq3nNU4Z1nz4TBFR0BdaHZX+L54qQNimpzJXoU8AwsnnP/4Rx/28oRukoxpuBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"db313ea97991b273f1aac591502cfa5eef2159f0dbc64c213f7855da2362f275","last_reissued_at":"2026-07-05T11:49:31.867817Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:49:31.867817Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Automatically Interpreting Millions of Features in Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.LG","authors_text":"Alex Mallen, Caden Juang, Gon\\c{c}alo Paulo, Nora Belrose","submitted_at":"2024-10-17T17:56:01Z","abstract_excerpt":"While the activations of neurons in deep neural networks usually do not have a simple human-understandable interpretation, sparse autoencoders (SAEs) can be used to transform these activations into a higher-dimensional latent space which may be more easily interpretable. However, these SAEs can have millions of distinct latent features, making it infeasible for humans to manually interpret each one. In this work, we build an open-source automated pipeline to generate and evaluate natural language explanations for SAE features using LLMs. We test our framework on SAEs of varying sizes, activati"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.13928","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.13928/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.13928","created_at":"2026-07-05T11:49:31.867874+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.13928v3","created_at":"2026-07-05T11:49:31.867874+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.13928","created_at":"2026-07-05T11:49:31.867874+00:00"},{"alias_kind":"pith_short_12","alias_value":"3MYT5KLZSGZH","created_at":"2026-07-05T11:49:31.867874+00:00"},{"alias_kind":"pith_short_16","alias_value":"3MYT5KLZSGZHH4NK","created_at":"2026-07-05T11:49:31.867874+00:00"},{"alias_kind":"pith_short_8","alias_value":"3MYT5KLZ","created_at":"2026-07-05T11:49:31.867874+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":20,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.22994","citing_title":"Do Sparse Autoencoders Learn Meaningful Concept Hierarchies?","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2606.11722","citing_title":"ICA Lens: Interpreting Language Models Without Training Another Dictionary","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2606.12360","citing_title":"Anatomy of Post-Training: Using Interpretability to Characterize Data and Shape the Learning Signal","ref_index":76,"is_internal_anchor":false},{"citing_arxiv_id":"2606.10029","citing_title":"Interpreting and Steering a Text-to-Speech Language Model with Sparse Autoencoders","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2606.08496","citing_title":"SAEExplainer: Interpreting SAE Features with Activation-Guided Preference Optimization","ref_index":39,"is_internal_anchor":false},{"citing_arxiv_id":"2605.14694","citing_title":"The Rate-Distortion-Polysemanticity Tradeoff in SAEs","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2606.21197","citing_title":"Extraction and Analysis of Multimodal Concepts in Vision Language Models through Sparse Autoencoders","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18789","citing_title":"Features have life history. And we should care","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18229","citing_title":"Are Sparse Autoencoder Benchmarks Reliable?","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2509.18127","citing_title":"Safe-SAIL: Towards a Fine-grained Safety Landscape of Large Language Models via Sparse Autoencoder Interpretation Framework","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2511.01680","citing_title":"Making Interpretable Discoveries from Unstructured Data: A High-Dimensional Multiple Hypothesis Testing Approach","ref_index":41,"is_internal_anchor":false},{"citing_arxiv_id":"2605.14192","citing_title":"Why Retrieval-Augmented Generation Fails: A Graph Perspective","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12874","citing_title":"Descriptive Collision in Sparse Autoencoder Auto-Interpretability: When One Explanation Describes Many Features","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2604.03436","citing_title":"MetaSAEs: Joint Training with a Decomposability Penalty Produces More Atomic Sparse Autoencoder Latents","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11920","citing_title":"Domain Restriction via Multi SAE Layer Transitions","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12412","citing_title":"Stories in Space: In-Context Learning Trajectories in Conceptual Belief Space","ref_index":168,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07922","citing_title":"Tree SAE: Learning Hierarchical Feature Structures in Sparse Autoencoders","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06494","citing_title":"From Token Lists to Graph Motifs: Weisfeiler-Lehman Analysis of Sparse Autoencoder Features","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2604.08846","citing_title":"Dictionary-Aligned Concept Control for Safeguarding Multimodal LLMs","ref_index":80,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07922","citing_title":"Tree SAE: Learning Hierarchical Feature Structures in Sparse Autoencoders","ref_index":19,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/3MYT5KLZSGZHH4NKYWIVALH2L3","json":"https://pith.science/pith/3MYT5KLZSGZHH4NKYWIVALH2L3.json","graph_json":"https://pith.science/api/pith-number/3MYT5KLZSGZHH4NKYWIVALH2L3/graph.json","events_json":"https://pith.science/api/pith-number/3MYT5KLZSGZHH4NKYWIVALH2L3/events.json","paper":"https://pith.science/paper/3MYT5KLZ"},"agent_actions":{"view_html":"https://pith.science/pith/3MYT5KLZSGZHH4NKYWIVALH2L3","download_json":"https://pith.science/pith/3MYT5KLZSGZHH4NKYWIVALH2L3.json","view_paper":"https://pith.science/paper/3MYT5KLZ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.13928&json=true","fetch_graph":"https://pith.science/api/pith-number/3MYT5KLZSGZHH4NKYWIVALH2L3/graph.json","fetch_events":"https://pith.science/api/pith-number/3MYT5KLZSGZHH4NKYWIVALH2L3/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/3MYT5KLZSGZHH4NKYWIVALH2L3/action/timestamp_anchor","attest_storage":"https://pith.science/pith/3MYT5KLZSGZHH4NKYWIVALH2L3/action/storage_attestation","attest_author":"https://pith.science/pith/3MYT5KLZSGZHH4NKYWIVALH2L3/action/author_attestation","sign_citation":"https://pith.science/pith/3MYT5KLZSGZHH4NKYWIVALH2L3/action/citation_signature","submit_replication":"https://pith.science/pith/3MYT5KLZSGZHH4NKYWIVALH2L3/action/replication_record"}},"created_at":"2026-07-05T11:49:31.867874+00:00","updated_at":"2026-07-05T11:49:31.867874+00:00"}