{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:T7U6H22HMMBEKAEEIWII57JB3K","short_pith_number":"pith:T7U6H22H","schema_version":"1.0","canonical_sha256":"9fe9e3eb47630245008445908efd21da89d4e7a6f1943f602dc48a90825f050d","source":{"kind":"arxiv","id":"2405.12522","version":1},"attestation_state":"computed","paper":{"title":"Sparse Autoencoders Enable Scalable and Reliable Circuit Identification in Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Charles O'Neill, Thang Bui","submitted_at":"2024-05-21T06:26:10Z","abstract_excerpt":"This paper introduces an efficient and robust method for discovering interpretable circuits in large language models using discrete sparse autoencoders. Our approach addresses key limitations of existing techniques, namely computational complexity and sensitivity to hyperparameters. We propose training sparse autoencoders on carefully designed positive and negative examples, where the model can only correctly predict the next token for the positive examples. We hypothesise that learned representations of attention head outputs will signal when a head is engaged in specific computations. By dis"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2405.12522","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-05-21T06:26:10Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"6f87de1edbb052407d13a32cdcbcd163804b1569d62b748e07bee1469d484943","abstract_canon_sha256":"869fda47c91fc0c07c1508cb10b06d300a464dc9ecb65f56246f1c9dc31a164c"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:21:21.260587Z","signature_b64":"O7JClYx0etHNX+VZacplOOWtHnQXXvSzZbJmihzkti/Jusy10EtOjvA4EqfZun9S7i5yZuJW6lAT6QkZr4QXDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"9fe9e3eb47630245008445908efd21da89d4e7a6f1943f602dc48a90825f050d","last_reissued_at":"2026-07-05T08:21:21.260183Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:21:21.260183Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Sparse Autoencoders Enable Scalable and Reliable Circuit Identification in Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Charles O'Neill, Thang Bui","submitted_at":"2024-05-21T06:26:10Z","abstract_excerpt":"This paper introduces an efficient and robust method for discovering interpretable circuits in large language models using discrete sparse autoencoders. Our approach addresses key limitations of existing techniques, namely computational complexity and sensitivity to hyperparameters. We propose training sparse autoencoders on carefully designed positive and negative examples, where the model can only correctly predict the next token for the positive examples. We hypothesise that learned representations of attention head outputs will signal when a head is engaged in specific computations. By dis"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2405.12522","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2405.12522/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2405.12522","created_at":"2026-07-05T08:21:21.260235+00:00"},{"alias_kind":"arxiv_version","alias_value":"2405.12522v1","created_at":"2026-07-05T08:21:21.260235+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2405.12522","created_at":"2026-07-05T08:21:21.260235+00:00"},{"alias_kind":"pith_short_12","alias_value":"T7U6H22HMMBE","created_at":"2026-07-05T08:21:21.260235+00:00"},{"alias_kind":"pith_short_16","alias_value":"T7U6H22HMMBEKAEE","created_at":"2026-07-05T08:21:21.260235+00:00"},{"alias_kind":"pith_short_8","alias_value":"T7U6H22H","created_at":"2026-07-05T08:21:21.260235+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.06267","citing_title":"Many Circuits, One Mechanism: Input Variation and Evaluation Granularity in Circuit Discovery","ref_index":76,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/T7U6H22HMMBEKAEEIWII57JB3K","json":"https://pith.science/pith/T7U6H22HMMBEKAEEIWII57JB3K.json","graph_json":"https://pith.science/api/pith-number/T7U6H22HMMBEKAEEIWII57JB3K/graph.json","events_json":"https://pith.science/api/pith-number/T7U6H22HMMBEKAEEIWII57JB3K/events.json","paper":"https://pith.science/paper/T7U6H22H"},"agent_actions":{"view_html":"https://pith.science/pith/T7U6H22HMMBEKAEEIWII57JB3K","download_json":"https://pith.science/pith/T7U6H22HMMBEKAEEIWII57JB3K.json","view_paper":"https://pith.science/paper/T7U6H22H","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2405.12522&json=true","fetch_graph":"https://pith.science/api/pith-number/T7U6H22HMMBEKAEEIWII57JB3K/graph.json","fetch_events":"https://pith.science/api/pith-number/T7U6H22HMMBEKAEEIWII57JB3K/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/T7U6H22HMMBEKAEEIWII57JB3K/action/timestamp_anchor","attest_storage":"https://pith.science/pith/T7U6H22HMMBEKAEEIWII57JB3K/action/storage_attestation","attest_author":"https://pith.science/pith/T7U6H22HMMBEKAEEIWII57JB3K/action/author_attestation","sign_citation":"https://pith.science/pith/T7U6H22HMMBEKAEEIWII57JB3K/action/citation_signature","submit_replication":"https://pith.science/pith/T7U6H22HMMBEKAEEIWII57JB3K/action/replication_record"}},"created_at":"2026-07-05T08:21:21.260235+00:00","updated_at":"2026-07-05T08:21:21.260235+00:00"}