{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:VO5FYNJN2GGPC2MNXBJFE6KXKH","short_pith_number":"pith:VO5FYNJN","schema_version":"1.0","canonical_sha256":"abba5c352dd18cf1698db85252795751c4f94352952563b1940832b4622cb37f","source":{"kind":"arxiv","id":"2502.19964","version":2},"attestation_state":"computed","paper":{"title":"Do Sparse Autoencoders Generalize? A Case Study of Answerability","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Fazl Barez, Lovis Heindrich, Philip Torr, Veronika Thost","submitted_at":"2025-02-27T10:45:25Z","abstract_excerpt":"Sparse autoencoders (SAEs) have emerged as a promising approach in language model interpretability, offering unsupervised extraction of sparse features. For interpretability methods to succeed, they must identify abstract features across domains, and these features can often manifest differently in each context. We examine this through \"answerability\" - a model's ability to recognize answerable questions. We extensively evaluate SAE feature generalization across diverse, partly self-constructed answerability datasets for Gemma 2 SAEs. Our analysis reveals that residual stream probes outperform"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.19964","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-02-27T10:45:25Z","cross_cats_sorted":[],"title_canon_sha256":"0d8ed54ad761a5580f22b8d424a849a5d86bebd178ca3af3e67cac5507767e06","abstract_canon_sha256":"b62428c1e5fc91cba5fc48b37cea89a74065f7481351d9719a406430634487b9"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T12:05:14.048971Z","signature_b64":"ayJswzv4n+EwoljO0Dcnl9Rq36BSUeJe5C09wCQl4P3bhjSKZCB4iaQQ4J1Y9tUAKxOFosKK+ZUcfghBJTpDDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"abba5c352dd18cf1698db85252795751c4f94352952563b1940832b4622cb37f","last_reissued_at":"2026-07-05T12:05:14.048404Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T12:05:14.048404Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Do Sparse Autoencoders Generalize? A Case Study of Answerability","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Fazl Barez, Lovis Heindrich, Philip Torr, Veronika Thost","submitted_at":"2025-02-27T10:45:25Z","abstract_excerpt":"Sparse autoencoders (SAEs) have emerged as a promising approach in language model interpretability, offering unsupervised extraction of sparse features. For interpretability methods to succeed, they must identify abstract features across domains, and these features can often manifest differently in each context. We examine this through \"answerability\" - a model's ability to recognize answerable questions. We extensively evaluate SAE feature generalization across diverse, partly self-constructed answerability datasets for Gemma 2 SAEs. Our analysis reveals that residual stream probes outperform"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.19964","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.19964/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.19964","created_at":"2026-07-05T12:05:14.048472+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.19964v2","created_at":"2026-07-05T12:05:14.048472+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.19964","created_at":"2026-07-05T12:05:14.048472+00:00"},{"alias_kind":"pith_short_12","alias_value":"VO5FYNJN2GGP","created_at":"2026-07-05T12:05:14.048472+00:00"},{"alias_kind":"pith_short_16","alias_value":"VO5FYNJN2GGPC2MN","created_at":"2026-07-05T12:05:14.048472+00:00"},{"alias_kind":"pith_short_8","alias_value":"VO5FYNJN","created_at":"2026-07-05T12:05:14.048472+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.17704","citing_title":"Toy Combinatorial Interpretability Models Reveal Lottery Tickets in Early Feature Space","ref_index":57,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/VO5FYNJN2GGPC2MNXBJFE6KXKH","json":"https://pith.science/pith/VO5FYNJN2GGPC2MNXBJFE6KXKH.json","graph_json":"https://pith.science/api/pith-number/VO5FYNJN2GGPC2MNXBJFE6KXKH/graph.json","events_json":"https://pith.science/api/pith-number/VO5FYNJN2GGPC2MNXBJFE6KXKH/events.json","paper":"https://pith.science/paper/VO5FYNJN"},"agent_actions":{"view_html":"https://pith.science/pith/VO5FYNJN2GGPC2MNXBJFE6KXKH","download_json":"https://pith.science/pith/VO5FYNJN2GGPC2MNXBJFE6KXKH.json","view_paper":"https://pith.science/paper/VO5FYNJN","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.19964&json=true","fetch_graph":"https://pith.science/api/pith-number/VO5FYNJN2GGPC2MNXBJFE6KXKH/graph.json","fetch_events":"https://pith.science/api/pith-number/VO5FYNJN2GGPC2MNXBJFE6KXKH/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/VO5FYNJN2GGPC2MNXBJFE6KXKH/action/timestamp_anchor","attest_storage":"https://pith.science/pith/VO5FYNJN2GGPC2MNXBJFE6KXKH/action/storage_attestation","attest_author":"https://pith.science/pith/VO5FYNJN2GGPC2MNXBJFE6KXKH/action/author_attestation","sign_citation":"https://pith.science/pith/VO5FYNJN2GGPC2MNXBJFE6KXKH/action/citation_signature","submit_replication":"https://pith.science/pith/VO5FYNJN2GGPC2MNXBJFE6KXKH/action/replication_record"}},"created_at":"2026-07-05T12:05:14.048472+00:00","updated_at":"2026-07-05T12:05:14.048472+00:00"}