{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:PEMHQRVPDG7CDZ3EHF2MK2VNRU","short_pith_number":"pith:PEMHQRVP","schema_version":"1.0","canonical_sha256":"79187846af19be21e7643974c56aad8d12da3314c2a8c6a5fdd63151f7937d64","source":{"kind":"arxiv","id":"2410.21331","version":1},"attestation_state":"computed","paper":{"title":"Beyond Interpretability: The Gains of Feature Monosemanticity on Model Robustness","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Jingyi Cui, Qi Lei, Qi Zhang, Stefanie Jegelka, Xiang Pan, Yifei Wang, Yisen Wang","submitted_at":"2024-10-27T18:03:20Z","abstract_excerpt":"Deep learning models often suffer from a lack of interpretability due to polysemanticity, where individual neurons are activated by multiple unrelated semantics, resulting in unclear attributions of model behavior. Recent advances in monosemanticity, where neurons correspond to consistent and distinct semantics, have significantly improved interpretability but are commonly believed to compromise accuracy. In this work, we challenge the prevailing belief of the accuracy-interpretability tradeoff, showing that monosemantic features not only enhance interpretability but also bring concrete gains "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.21331","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-10-27T18:03:20Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"29740f5a9f7d0d756d3d6a2c5088fff1b500838eeaf06b2edf436743b1f8b5f9","abstract_canon_sha256":"b80b9661380f57efd81bdb57c6204563097cd66b223689840e5cf48f61deb241"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:27:26.914838Z","signature_b64":"4qKCcEQof6W6CZKxayXD5qIGHzqVgJ+U+aclplQgnnHIgSVvF1OptY25f0JIXv0EE2ZcJ8Of3KqEiG2lXLqMBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"79187846af19be21e7643974c56aad8d12da3314c2a8c6a5fdd63151f7937d64","last_reissued_at":"2026-07-05T09:27:26.914354Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:27:26.914354Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Beyond Interpretability: The Gains of Feature Monosemanticity on Model Robustness","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Jingyi Cui, Qi Lei, Qi Zhang, Stefanie Jegelka, Xiang Pan, Yifei Wang, Yisen Wang","submitted_at":"2024-10-27T18:03:20Z","abstract_excerpt":"Deep learning models often suffer from a lack of interpretability due to polysemanticity, where individual neurons are activated by multiple unrelated semantics, resulting in unclear attributions of model behavior. Recent advances in monosemanticity, where neurons correspond to consistent and distinct semantics, have significantly improved interpretability but are commonly believed to compromise accuracy. In this work, we challenge the prevailing belief of the accuracy-interpretability tradeoff, showing that monosemantic features not only enhance interpretability but also bring concrete gains "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.21331","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.21331/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.21331","created_at":"2026-07-05T09:27:26.914414+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.21331v1","created_at":"2026-07-05T09:27:26.914414+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.21331","created_at":"2026-07-05T09:27:26.914414+00:00"},{"alias_kind":"pith_short_12","alias_value":"PEMHQRVPDG7C","created_at":"2026-07-05T09:27:26.914414+00:00"},{"alias_kind":"pith_short_16","alias_value":"PEMHQRVPDG7CDZ3E","created_at":"2026-07-05T09:27:26.914414+00:00"},{"alias_kind":"pith_short_8","alias_value":"PEMHQRVP","created_at":"2026-07-05T09:27:26.914414+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.07714","citing_title":"Beyond Accuracy: Interpreting Topic Representation in Suicide Ideation Detection Models","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06610","citing_title":"SoftSAE: Dynamic Top-K Selection for Adaptive Sparse Autoencoders","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06610","citing_title":"SoftSAE: Dynamic Top-K Selection for Adaptive Sparse Autoencoders","ref_index":2,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/PEMHQRVPDG7CDZ3EHF2MK2VNRU","json":"https://pith.science/pith/PEMHQRVPDG7CDZ3EHF2MK2VNRU.json","graph_json":"https://pith.science/api/pith-number/PEMHQRVPDG7CDZ3EHF2MK2VNRU/graph.json","events_json":"https://pith.science/api/pith-number/PEMHQRVPDG7CDZ3EHF2MK2VNRU/events.json","paper":"https://pith.science/paper/PEMHQRVP"},"agent_actions":{"view_html":"https://pith.science/pith/PEMHQRVPDG7CDZ3EHF2MK2VNRU","download_json":"https://pith.science/pith/PEMHQRVPDG7CDZ3EHF2MK2VNRU.json","view_paper":"https://pith.science/paper/PEMHQRVP","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.21331&json=true","fetch_graph":"https://pith.science/api/pith-number/PEMHQRVPDG7CDZ3EHF2MK2VNRU/graph.json","fetch_events":"https://pith.science/api/pith-number/PEMHQRVPDG7CDZ3EHF2MK2VNRU/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/PEMHQRVPDG7CDZ3EHF2MK2VNRU/action/timestamp_anchor","attest_storage":"https://pith.science/pith/PEMHQRVPDG7CDZ3EHF2MK2VNRU/action/storage_attestation","attest_author":"https://pith.science/pith/PEMHQRVPDG7CDZ3EHF2MK2VNRU/action/author_attestation","sign_citation":"https://pith.science/pith/PEMHQRVPDG7CDZ3EHF2MK2VNRU/action/citation_signature","submit_replication":"https://pith.science/pith/PEMHQRVPDG7CDZ3EHF2MK2VNRU/action/replication_record"}},"created_at":"2026-07-05T09:27:26.914414+00:00","updated_at":"2026-07-05T09:27:26.914414+00:00"}