{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:VYV7BBXJ5RCDSBZI7A4DJNV34E","short_pith_number":"pith:VYV7BBXJ","schema_version":"1.0","canonical_sha256":"ae2bf086e9ec44390728f83834b6bbe13ad9ca32c673eb326238888b3825ba1b","source":{"kind":"arxiv","id":"2503.18878","version":2},"attestation_state":"computed","paper":{"title":"I Have Covered All the Bases Here: Interpreting Reasoning Features in Large Language Models via Sparse Autoencoders","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Alexey Dontsov, Andrey Galichin, Anton Razzhigaev, Elena Tutubalina, Ivan Oseledets, Oleg Y. Rogov, Polina Druzhinina","submitted_at":"2025-03-24T16:54:26Z","abstract_excerpt":"Recent LLMs like DeepSeek-R1 have demonstrated state-of-the-art performance by integrating deep thinking and complex reasoning during generation. However, the internal mechanisms behind these reasoning processes remain unexplored. We observe reasoning LLMs consistently use vocabulary associated with human reasoning processes. We hypothesize these words correspond to specific reasoning moments within the models' internal mechanisms. To test this hypothesis, we employ Sparse Autoencoders (SAEs), a technique for sparse decomposition of neural network activations into human-interpretable features."},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2503.18878","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.CL","submitted_at":"2025-03-24T16:54:26Z","cross_cats_sorted":[],"title_canon_sha256":"8e3a4b9a79c02e3fc4f3d101604bfd4d3fdb12b0a52b1f58ad611efb49eaa156","abstract_canon_sha256":"c7f5b38063f4dbdc71755ee82ed3c098ff177d146939f3813daaa4ec376f90ea"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:49:08.594203Z","signature_b64":"wn+wRtdu6CLtgs1Qt7xMZf39d1meDNb7pDFCy+AGpctD3OHcCCnQFG7aVRMQClDBq/DfZ0pR/1QrYmkCnHEGBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ae2bf086e9ec44390728f83834b6bbe13ad9ca32c673eb326238888b3825ba1b","last_reissued_at":"2026-07-05T11:49:08.593613Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:49:08.593613Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"I Have Covered All the Bases Here: Interpreting Reasoning Features in Large Language Models via Sparse Autoencoders","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Alexey Dontsov, Andrey Galichin, Anton Razzhigaev, Elena Tutubalina, Ivan Oseledets, Oleg Y. Rogov, Polina Druzhinina","submitted_at":"2025-03-24T16:54:26Z","abstract_excerpt":"Recent LLMs like DeepSeek-R1 have demonstrated state-of-the-art performance by integrating deep thinking and complex reasoning during generation. However, the internal mechanisms behind these reasoning processes remain unexplored. We observe reasoning LLMs consistently use vocabulary associated with human reasoning processes. We hypothesize these words correspond to specific reasoning moments within the models' internal mechanisms. To test this hypothesis, we employ Sparse Autoencoders (SAEs), a technique for sparse decomposition of neural network activations into human-interpretable features."},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2503.18878","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2503.18878/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2503.18878","created_at":"2026-07-05T11:49:08.593676+00:00"},{"alias_kind":"arxiv_version","alias_value":"2503.18878v2","created_at":"2026-07-05T11:49:08.593676+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2503.18878","created_at":"2026-07-05T11:49:08.593676+00:00"},{"alias_kind":"pith_short_12","alias_value":"VYV7BBXJ5RCD","created_at":"2026-07-05T11:49:08.593676+00:00"},{"alias_kind":"pith_short_16","alias_value":"VYV7BBXJ5RCDSBZI","created_at":"2026-07-05T11:49:08.593676+00:00"},{"alias_kind":"pith_short_8","alias_value":"VYV7BBXJ","created_at":"2026-07-05T11:49:08.593676+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":7,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.11470","citing_title":"The Periodic Table of LLM Reasoning: A Structured Survey of Reasoning Paradigms, Methods, and Failure Modes","ref_index":65,"is_internal_anchor":false},{"citing_arxiv_id":"2606.06188","citing_title":"The Tell-Tale Norm: $\\ell_2$ Magnitude as a Signal for Reasoning Dynamics in Large Language Models","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2606.06840","citing_title":"Characterize Then Distill: Mechanistic Reasoning in Large Output Spaces","ref_index":51,"is_internal_anchor":false},{"citing_arxiv_id":"2605.23040","citing_title":"Steered Generation via Gradient-Based Optimization on Sparse Query Features","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2601.14004","citing_title":"Locate, Steer, and Improve: A Practical Survey of Actionable Mechanistic Interpretability in Large Language Models","ref_index":76,"is_internal_anchor":false},{"citing_arxiv_id":"2604.25011","citing_title":"Why Does Reinforcement Learning Generalize? A Feature-Level Mechanistic Study of Post-Training in Large Language Models","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2604.17761","citing_title":"Contrastive Attribution in the Wild: An Interpretability Analysis of LLM Failures on Realistic Benchmarks","ref_index":25,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/VYV7BBXJ5RCDSBZI7A4DJNV34E","json":"https://pith.science/pith/VYV7BBXJ5RCDSBZI7A4DJNV34E.json","graph_json":"https://pith.science/api/pith-number/VYV7BBXJ5RCDSBZI7A4DJNV34E/graph.json","events_json":"https://pith.science/api/pith-number/VYV7BBXJ5RCDSBZI7A4DJNV34E/events.json","paper":"https://pith.science/paper/VYV7BBXJ"},"agent_actions":{"view_html":"https://pith.science/pith/VYV7BBXJ5RCDSBZI7A4DJNV34E","download_json":"https://pith.science/pith/VYV7BBXJ5RCDSBZI7A4DJNV34E.json","view_paper":"https://pith.science/paper/VYV7BBXJ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2503.18878&json=true","fetch_graph":"https://pith.science/api/pith-number/VYV7BBXJ5RCDSBZI7A4DJNV34E/graph.json","fetch_events":"https://pith.science/api/pith-number/VYV7BBXJ5RCDSBZI7A4DJNV34E/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/VYV7BBXJ5RCDSBZI7A4DJNV34E/action/timestamp_anchor","attest_storage":"https://pith.science/pith/VYV7BBXJ5RCDSBZI7A4DJNV34E/action/storage_attestation","attest_author":"https://pith.science/pith/VYV7BBXJ5RCDSBZI7A4DJNV34E/action/author_attestation","sign_citation":"https://pith.science/pith/VYV7BBXJ5RCDSBZI7A4DJNV34E/action/citation_signature","submit_replication":"https://pith.science/pith/VYV7BBXJ5RCDSBZI7A4DJNV34E/action/replication_record"}},"created_at":"2026-07-05T11:49:08.593676+00:00","updated_at":"2026-07-05T11:49:08.593676+00:00"}