{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:25WXSAZFDS25U4WK2SCVGV22GZ","short_pith_number":"pith:25WXSAZF","schema_version":"1.0","canonical_sha256":"d76d7903251cb5da72cad48553575a364fc8b8e96bff399eea6d9295f4a6a60a","source":{"kind":"arxiv","id":"2411.01220","version":2},"attestation_state":"computed","paper":{"title":"Enhancing Neural Network Interpretability with Feature-Aligned Sparse Autoencoders","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Alasdair Paren, David Krueger, Fazl Barez, Luke Marks","submitted_at":"2024-11-02T11:42:23Z","abstract_excerpt":"Sparse Autoencoders (SAEs) have shown promise in improving the interpretability of neural network activations, but can learn features that are not features of the input, limiting their effectiveness. We propose \\textsc{Mutual Feature Regularization} \\textbf{(MFR)}, a regularization technique for improving feature learning by encouraging SAEs trained in parallel to learn similar features. We motivate \\textsc{MFR} by showing that features learned by multiple SAEs are more likely to correlate with features of the input. By training on synthetic data with known features of the input, we show that "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2411.01220","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-11-02T11:42:23Z","cross_cats_sorted":[],"title_canon_sha256":"f9e033a8ea19fc31b76deb2b965527566e7ec4e2538043476daace89a9be1585","abstract_canon_sha256":"502b31a76d456c739553d3a154e5943db9153b1c27b9e35c58fef4bd8a49f770"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:31:50.819752Z","signature_b64":"rTwv/tR9Tuk7OEa/XK+4Q0q+39PrmIM49GVWPdWElSxkxFICzEcqKtIniH3CjRzWTAopTmdhyZx8ohLGX8FWAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"d76d7903251cb5da72cad48553575a364fc8b8e96bff399eea6d9295f4a6a60a","last_reissued_at":"2026-07-05T09:31:50.819246Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:31:50.819246Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Enhancing Neural Network Interpretability with Feature-Aligned Sparse Autoencoders","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Alasdair Paren, David Krueger, Fazl Barez, Luke Marks","submitted_at":"2024-11-02T11:42:23Z","abstract_excerpt":"Sparse Autoencoders (SAEs) have shown promise in improving the interpretability of neural network activations, but can learn features that are not features of the input, limiting their effectiveness. We propose \\textsc{Mutual Feature Regularization} \\textbf{(MFR)}, a regularization technique for improving feature learning by encouraging SAEs trained in parallel to learn similar features. We motivate \\textsc{MFR} by showing that features learned by multiple SAEs are more likely to correlate with features of the input. By training on synthetic data with known features of the input, we show that "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2411.01220","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2411.01220/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2411.01220","created_at":"2026-07-05T09:31:50.819302+00:00"},{"alias_kind":"arxiv_version","alias_value":"2411.01220v2","created_at":"2026-07-05T09:31:50.819302+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2411.01220","created_at":"2026-07-05T09:31:50.819302+00:00"},{"alias_kind":"pith_short_12","alias_value":"25WXSAZFDS25","created_at":"2026-07-05T09:31:50.819302+00:00"},{"alias_kind":"pith_short_16","alias_value":"25WXSAZFDS25U4WK","created_at":"2026-07-05T09:31:50.819302+00:00"},{"alias_kind":"pith_short_8","alias_value":"25WXSAZF","created_at":"2026-07-05T09:31:50.819302+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":7,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.08499","citing_title":"Cross-seed explainability using Procrustes-conditioned Joint End-to-end Top-K Sparse Autoencoders","ref_index":8,"is_internal_anchor":true},{"citing_arxiv_id":"2606.09653","citing_title":"A Unifying Framework for Concept-Based Representational Similarity","ref_index":72,"is_internal_anchor":false},{"citing_arxiv_id":"2606.03002","citing_title":"Perplexity Can Miss SAE Feature Damage Under Quantization","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18629","citing_title":"Aligned Training: A Parameter-Free Method to Improve Feature Quality and Stability of Sparse Autoencoders (SAE)","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18629","citing_title":"Aligned Training: A Parameter-Free Method to Improve Feature Quality and Stability of Sparse Autoencoders (SAE)","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2604.04946","citing_title":"Sparse Autoencoders as a Steering Basis for Phase Synchronization in Graph-Based CFD Surrogates","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2604.14925","citing_title":"Improving Sparse Autoencoder with Dynamic Attention","ref_index":38,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/25WXSAZFDS25U4WK2SCVGV22GZ","json":"https://pith.science/pith/25WXSAZFDS25U4WK2SCVGV22GZ.json","graph_json":"https://pith.science/api/pith-number/25WXSAZFDS25U4WK2SCVGV22GZ/graph.json","events_json":"https://pith.science/api/pith-number/25WXSAZFDS25U4WK2SCVGV22GZ/events.json","paper":"https://pith.science/paper/25WXSAZF"},"agent_actions":{"view_html":"https://pith.science/pith/25WXSAZFDS25U4WK2SCVGV22GZ","download_json":"https://pith.science/pith/25WXSAZFDS25U4WK2SCVGV22GZ.json","view_paper":"https://pith.science/paper/25WXSAZF","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2411.01220&json=true","fetch_graph":"https://pith.science/api/pith-number/25WXSAZFDS25U4WK2SCVGV22GZ/graph.json","fetch_events":"https://pith.science/api/pith-number/25WXSAZFDS25U4WK2SCVGV22GZ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/25WXSAZFDS25U4WK2SCVGV22GZ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/25WXSAZFDS25U4WK2SCVGV22GZ/action/storage_attestation","attest_author":"https://pith.science/pith/25WXSAZFDS25U4WK2SCVGV22GZ/action/author_attestation","sign_citation":"https://pith.science/pith/25WXSAZFDS25U4WK2SCVGV22GZ/action/citation_signature","submit_replication":"https://pith.science/pith/25WXSAZFDS25U4WK2SCVGV22GZ/action/replication_record"}},"created_at":"2026-07-05T09:31:50.819302+00:00","updated_at":"2026-07-05T09:31:50.819302+00:00"}