{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:QARUC5BNG6SNTNH2GJY2MISDLH","short_pith_number":"pith:QARUC5BN","schema_version":"1.0","canonical_sha256":"802341742d37a4d9b4fa3271a6224359f5c2b66b7eda1807c33c7a39cb25d88d","source":{"kind":"arxiv","id":"2410.11179","version":1},"attestation_state":"computed","paper":{"title":"Interpretability as Compression: Reconsidering SAE Explanations of Neural Activations with MDL-SAEs","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.IT","math.IT"],"primary_cat":"cs.LG","authors_text":"Kola Ayonrinde, Lee Sharkey, Michael T. Pearce","submitted_at":"2024-10-15T01:38:03Z","abstract_excerpt":"Sparse Autoencoders (SAEs) have emerged as a useful tool for interpreting the internal representations of neural networks. However, naively optimising SAEs for reconstruction loss and sparsity results in a preference for SAEs that are extremely wide and sparse. We present an information-theoretic framework for interpreting SAEs as lossy compression algorithms for communicating explanations of neural activations. We appeal to the Minimal Description Length (MDL) principle to motivate explanations of activations which are both accurate and concise. We further argue that interpretable SAEs requir"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.11179","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-10-15T01:38:03Z","cross_cats_sorted":["cs.AI","cs.IT","math.IT"],"title_canon_sha256":"dc3f1bc4f0fce8a09e99b6e5ef78db33f3c1f39f45bd1ae972fbd21c4094dd23","abstract_canon_sha256":"485641402d50a0812c9891434033630aae6eacfa202da881b67d4a32228a39fc"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:20:30.394130Z","signature_b64":"8LtnB3n02zmzV8vS31RAuN/2ac+N8rw3Dh+ZOM1wThNOq9xGMQsi6Ee3TKYZ/1l2Crw9Wj0L4/40a6HIDw3kBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"802341742d37a4d9b4fa3271a6224359f5c2b66b7eda1807c33c7a39cb25d88d","last_reissued_at":"2026-07-05T09:20:30.393734Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:20:30.393734Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Interpretability as Compression: Reconsidering SAE Explanations of Neural Activations with MDL-SAEs","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.IT","math.IT"],"primary_cat":"cs.LG","authors_text":"Kola Ayonrinde, Lee Sharkey, Michael T. Pearce","submitted_at":"2024-10-15T01:38:03Z","abstract_excerpt":"Sparse Autoencoders (SAEs) have emerged as a useful tool for interpreting the internal representations of neural networks. However, naively optimising SAEs for reconstruction loss and sparsity results in a preference for SAEs that are extremely wide and sparse. We present an information-theoretic framework for interpreting SAEs as lossy compression algorithms for communicating explanations of neural activations. We appeal to the Minimal Description Length (MDL) principle to motivate explanations of activations which are both accurate and concise. We further argue that interpretable SAEs requir"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.11179","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.11179/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.11179","created_at":"2026-07-05T09:20:30.393798+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.11179v1","created_at":"2026-07-05T09:20:30.393798+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.11179","created_at":"2026-07-05T09:20:30.393798+00:00"},{"alias_kind":"pith_short_12","alias_value":"QARUC5BNG6SN","created_at":"2026-07-05T09:20:30.393798+00:00"},{"alias_kind":"pith_short_16","alias_value":"QARUC5BNG6SNTNH2","created_at":"2026-07-05T09:20:30.393798+00:00"},{"alias_kind":"pith_short_8","alias_value":"QARUC5BN","created_at":"2026-07-05T09:20:30.393798+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":8,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.25234","citing_title":"Structuring Sparsity: Block-Sparse Featurizers Capture Visual Concept Manifolds","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08934","citing_title":"From Mechanistic to Compositional Interpretability","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2605.14694","citing_title":"The Rate-Distortion-Polysemanticity Tradeoff in SAEs","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12874","citing_title":"Descriptive Collision in Sparse Autoencoder Auto-Interpretability: When One Explanation Describes Many Features","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08934","citing_title":"From Mechanistic to Compositional Interpretability","ref_index":160,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07922","citing_title":"Tree SAE: Learning Hierarchical Feature Structures in Sparse Autoencoders","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08934","citing_title":"From Mechanistic to Compositional Interpretability","ref_index":154,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07922","citing_title":"Tree SAE: Learning Hierarchical Feature Structures in Sparse Autoencoders","ref_index":1,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/QARUC5BNG6SNTNH2GJY2MISDLH","json":"https://pith.science/pith/QARUC5BNG6SNTNH2GJY2MISDLH.json","graph_json":"https://pith.science/api/pith-number/QARUC5BNG6SNTNH2GJY2MISDLH/graph.json","events_json":"https://pith.science/api/pith-number/QARUC5BNG6SNTNH2GJY2MISDLH/events.json","paper":"https://pith.science/paper/QARUC5BN"},"agent_actions":{"view_html":"https://pith.science/pith/QARUC5BNG6SNTNH2GJY2MISDLH","download_json":"https://pith.science/pith/QARUC5BNG6SNTNH2GJY2MISDLH.json","view_paper":"https://pith.science/paper/QARUC5BN","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.11179&json=true","fetch_graph":"https://pith.science/api/pith-number/QARUC5BNG6SNTNH2GJY2MISDLH/graph.json","fetch_events":"https://pith.science/api/pith-number/QARUC5BNG6SNTNH2GJY2MISDLH/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/QARUC5BNG6SNTNH2GJY2MISDLH/action/timestamp_anchor","attest_storage":"https://pith.science/pith/QARUC5BNG6SNTNH2GJY2MISDLH/action/storage_attestation","attest_author":"https://pith.science/pith/QARUC5BNG6SNTNH2GJY2MISDLH/action/author_attestation","sign_citation":"https://pith.science/pith/QARUC5BNG6SNTNH2GJY2MISDLH/action/citation_signature","submit_replication":"https://pith.science/pith/QARUC5BNG6SNTNH2GJY2MISDLH/action/replication_record"}},"created_at":"2026-07-05T09:20:30.393798+00:00","updated_at":"2026-07-05T09:20:30.393798+00:00"}