{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:HR6OHQBWAVNHFM6GVJGNUSEO2Z","short_pith_number":"pith:HR6OHQBW","schema_version":"1.0","canonical_sha256":"3c7ce3c036055a72b3c6aa4cda488ed6759533faa5ed5e7757430f369b78ba65","source":{"kind":"arxiv","id":"2301.04709","version":4},"attestation_state":"computed","paper":{"title":"Causal Abstraction: A Theoretical Foundation for Mechanistic Interpretability","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Amir Zur, Aryaman Arora, Atticus Geiger, Christopher Potts, Duligur Ibeling, Jing Huang, Maheep Chaudhary, Noah Goodman, Sonakshi Chauhan, Thomas Icard, Zhengxuan Wu","submitted_at":"2023-01-11T20:42:41Z","abstract_excerpt":"Causal abstraction provides a theoretical foundation for mechanistic interpretability, the field concerned with providing intelligible algorithms that are faithful simplifications of the known, but opaque low-level details of black box AI models. Our contributions are (1) generalizing the theory of causal abstraction from mechanism replacement (i.e., hard and soft interventions) to arbitrary mechanism transformation (i.e., functionals from old mechanisms to new mechanisms), (2) providing a flexible, yet precise formalization for the core concepts of polysemantic neurons, the linear representat"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2301.04709","kind":"arxiv","version":4},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2023-01-11T20:42:41Z","cross_cats_sorted":[],"title_canon_sha256":"2ae8012d97e5e3da66f1bec2596c2090e3fa4337d232af704f0249cabab35f51","abstract_canon_sha256":"27e45306f56d208c67368389926496f5beffb0461e04321085eecbcb84242594"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:00:34.439165Z","signature_b64":"Kb10uqSImULwSmEfjtakw/XxafesMXHQ+UqLNjq4erDgnSDl+JZTdBf9yXWZ4lnmKE77zOVD4SZxNuhQh0+RDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"3c7ce3c036055a72b3c6aa4cda488ed6759533faa5ed5e7757430f369b78ba65","last_reissued_at":"2026-07-05T11:00:34.438670Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:00:34.438670Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Causal Abstraction: A Theoretical Foundation for Mechanistic Interpretability","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Amir Zur, Aryaman Arora, Atticus Geiger, Christopher Potts, Duligur Ibeling, Jing Huang, Maheep Chaudhary, Noah Goodman, Sonakshi Chauhan, Thomas Icard, Zhengxuan Wu","submitted_at":"2023-01-11T20:42:41Z","abstract_excerpt":"Causal abstraction provides a theoretical foundation for mechanistic interpretability, the field concerned with providing intelligible algorithms that are faithful simplifications of the known, but opaque low-level details of black box AI models. Our contributions are (1) generalizing the theory of causal abstraction from mechanism replacement (i.e., hard and soft interventions) to arbitrary mechanism transformation (i.e., functionals from old mechanisms to new mechanisms), (2) providing a flexible, yet precise formalization for the core concepts of polysemantic neurons, the linear representat"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2301.04709","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2301.04709/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2301.04709","created_at":"2026-07-05T11:00:34.438735+00:00"},{"alias_kind":"arxiv_version","alias_value":"2301.04709v4","created_at":"2026-07-05T11:00:34.438735+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2301.04709","created_at":"2026-07-05T11:00:34.438735+00:00"},{"alias_kind":"pith_short_12","alias_value":"HR6OHQBWAVNH","created_at":"2026-07-05T11:00:34.438735+00:00"},{"alias_kind":"pith_short_16","alias_value":"HR6OHQBWAVNHFM6G","created_at":"2026-07-05T11:00:34.438735+00:00"},{"alias_kind":"pith_short_8","alias_value":"HR6OHQBW","created_at":"2026-07-05T11:00:34.438735+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":17,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.25657","citing_title":"Steering Vision-Language Models with Joint Sparse Autoencoders","ref_index":63,"is_internal_anchor":false},{"citing_arxiv_id":"2606.19741","citing_title":"Interpreting Neural Combinatorial Optimization via Evolving Programmatic Bottlenecks","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2606.10877","citing_title":"XtrAIn: Training-Guided Occlusion for Feature Attribution","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2606.08044","citing_title":"When Behavioral Safety Evaluation Fails: A Representation-Level Perspective","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2606.05194","citing_title":"Temporal Preference Concepts and their Functions in a Large Language Model","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2606.29657","citing_title":"Safety from Honesty in a Disinterested AI Predictor","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2605.25891","citing_title":"Causal Tongue-Tie: LLMs Can Encode Causal Direction, But Their Yes/No Outputs Fail to Express","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2605.26431","citing_title":"Probing LLMs for Syntactic Structure Beyond Universal Dependencies: A Minimalist Phase Account in English","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2606.06840","citing_title":"Characterize Then Distill: Mechanistic Reasoning in Large Output Spaces","ref_index":86,"is_internal_anchor":false},{"citing_arxiv_id":"2605.15328","citing_title":"From Weight Perturbation to Feature Attribution for Explaining Fully Connected Neural Networks","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2304.05969","citing_title":"Localizing Model Behavior with Path Patching","ref_index":42,"is_internal_anchor":false},{"citing_arxiv_id":"2310.15154","citing_title":"Linear Representations of Sentiment in Large Language Models","ref_index":90,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12809","citing_title":"Correcting Influence: Unboxing LLM Outputs with Orthogonal Latent Spaces","ref_index":76,"is_internal_anchor":false},{"citing_arxiv_id":"2403.19647","citing_title":"Sparse Feature Circuits: Discovering and Editing Interpretable Causal Graphs in Language Models","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08934","citing_title":"From Mechanistic to Compositional Interpretability","ref_index":147,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01164","citing_title":"LLMs Should Not Yet Be Credited with Decision Explanation","ref_index":38,"is_internal_anchor":false},{"citing_arxiv_id":"2605.02234","citing_title":"Bucketing the Good Apples: A Method for Diagnosing and Improving Causal Abstraction","ref_index":7,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/HR6OHQBWAVNHFM6GVJGNUSEO2Z","json":"https://pith.science/pith/HR6OHQBWAVNHFM6GVJGNUSEO2Z.json","graph_json":"https://pith.science/api/pith-number/HR6OHQBWAVNHFM6GVJGNUSEO2Z/graph.json","events_json":"https://pith.science/api/pith-number/HR6OHQBWAVNHFM6GVJGNUSEO2Z/events.json","paper":"https://pith.science/paper/HR6OHQBW"},"agent_actions":{"view_html":"https://pith.science/pith/HR6OHQBWAVNHFM6GVJGNUSEO2Z","download_json":"https://pith.science/pith/HR6OHQBWAVNHFM6GVJGNUSEO2Z.json","view_paper":"https://pith.science/paper/HR6OHQBW","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2301.04709&json=true","fetch_graph":"https://pith.science/api/pith-number/HR6OHQBWAVNHFM6GVJGNUSEO2Z/graph.json","fetch_events":"https://pith.science/api/pith-number/HR6OHQBWAVNHFM6GVJGNUSEO2Z/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/HR6OHQBWAVNHFM6GVJGNUSEO2Z/action/timestamp_anchor","attest_storage":"https://pith.science/pith/HR6OHQBWAVNHFM6GVJGNUSEO2Z/action/storage_attestation","attest_author":"https://pith.science/pith/HR6OHQBWAVNHFM6GVJGNUSEO2Z/action/author_attestation","sign_citation":"https://pith.science/pith/HR6OHQBWAVNHFM6GVJGNUSEO2Z/action/citation_signature","submit_replication":"https://pith.science/pith/HR6OHQBWAVNHFM6GVJGNUSEO2Z/action/replication_record"}},"created_at":"2026-07-05T11:00:34.438735+00:00","updated_at":"2026-07-05T11:00:34.438735+00:00"}