{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:5IEDEVPE6B6N6SNHXR5KORBY7U","short_pith_number":"pith:5IEDEVPE","schema_version":"1.0","canonical_sha256":"ea083255e4f07cdf49a7bc7aa74438fd3c3862f84f1f49c4846a0711dc2baa5e","source":{"kind":"arxiv","id":"2106.02997","version":2},"attestation_state":"computed","paper":{"title":"Causal Abstractions of Neural Networks","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.AI","authors_text":"Atticus Geiger, Christopher Potts, Hanson Lu, Thomas Icard","submitted_at":"2021-06-06T01:07:43Z","abstract_excerpt":"Structural analysis methods (e.g., probing and feature attribution) are increasingly important tools for neural network analysis. We propose a new structural analysis method grounded in a formal theory of causal abstraction that provides rich characterizations of model-internal representations and their roles in input/output behavior. In this method, neural representations are aligned with variables in interpretable causal models, and then interchange interventions are used to experimentally verify that the neural representations have the causal properties of their aligned variables. We apply "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2106.02997","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2021-06-06T01:07:43Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"05b38282973c4f5ed06ec1bb75875a6d2e4c09139ec089d25204a27ee942b416","abstract_canon_sha256":"96472a1507937d535f8ba2c0845dd73bf86a9e4af9f5e97139afcab85734e139"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T03:26:17.574043Z","signature_b64":"LpRuJtyzEE0Dc8bMNo5fmG4rVloP6/gX+k/cw33+P+oEjzTQpa7aqLMZ8WqZ9h4gPz4Y5i6jSNl2kHKSuIjQAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ea083255e4f07cdf49a7bc7aa74438fd3c3862f84f1f49c4846a0711dc2baa5e","last_reissued_at":"2026-07-05T03:26:17.573470Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T03:26:17.573470Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Causal Abstractions of Neural Networks","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.AI","authors_text":"Atticus Geiger, Christopher Potts, Hanson Lu, Thomas Icard","submitted_at":"2021-06-06T01:07:43Z","abstract_excerpt":"Structural analysis methods (e.g., probing and feature attribution) are increasingly important tools for neural network analysis. We propose a new structural analysis method grounded in a formal theory of causal abstraction that provides rich characterizations of model-internal representations and their roles in input/output behavior. In this method, neural representations are aligned with variables in interpretable causal models, and then interchange interventions are used to experimentally verify that the neural representations have the causal properties of their aligned variables. We apply "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2106.02997","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2106.02997/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2106.02997","created_at":"2026-07-05T03:26:17.573544+00:00"},{"alias_kind":"arxiv_version","alias_value":"2106.02997v2","created_at":"2026-07-05T03:26:17.573544+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2106.02997","created_at":"2026-07-05T03:26:17.573544+00:00"},{"alias_kind":"pith_short_12","alias_value":"5IEDEVPE6B6N","created_at":"2026-07-05T03:26:17.573544+00:00"},{"alias_kind":"pith_short_16","alias_value":"5IEDEVPE6B6N6SNH","created_at":"2026-07-05T03:26:17.573544+00:00"},{"alias_kind":"pith_short_8","alias_value":"5IEDEVPE","created_at":"2026-07-05T03:26:17.573544+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2607.01336","citing_title":"Mechanistic Interpretability and Causal Feature Steering of Neural Quantum States via Sparse Autoencoders","ref_index":60,"is_internal_anchor":false},{"citing_arxiv_id":"2607.00267","citing_title":"Validating Causal Abstraction Metrics on Simulated Complex Systems","ref_index":274,"is_internal_anchor":false},{"citing_arxiv_id":"2112.07874","citing_title":"Linguistic Frameworks Go Toe-to-Toe at Neuro-Symbolic Language Modeling","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2404.15255","citing_title":"How to use and interpret activation patching","ref_index":6,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/5IEDEVPE6B6N6SNHXR5KORBY7U","json":"https://pith.science/pith/5IEDEVPE6B6N6SNHXR5KORBY7U.json","graph_json":"https://pith.science/api/pith-number/5IEDEVPE6B6N6SNHXR5KORBY7U/graph.json","events_json":"https://pith.science/api/pith-number/5IEDEVPE6B6N6SNHXR5KORBY7U/events.json","paper":"https://pith.science/paper/5IEDEVPE"},"agent_actions":{"view_html":"https://pith.science/pith/5IEDEVPE6B6N6SNHXR5KORBY7U","download_json":"https://pith.science/pith/5IEDEVPE6B6N6SNHXR5KORBY7U.json","view_paper":"https://pith.science/paper/5IEDEVPE","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2106.02997&json=true","fetch_graph":"https://pith.science/api/pith-number/5IEDEVPE6B6N6SNHXR5KORBY7U/graph.json","fetch_events":"https://pith.science/api/pith-number/5IEDEVPE6B6N6SNHXR5KORBY7U/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/5IEDEVPE6B6N6SNHXR5KORBY7U/action/timestamp_anchor","attest_storage":"https://pith.science/pith/5IEDEVPE6B6N6SNHXR5KORBY7U/action/storage_attestation","attest_author":"https://pith.science/pith/5IEDEVPE6B6N6SNHXR5KORBY7U/action/author_attestation","sign_citation":"https://pith.science/pith/5IEDEVPE6B6N6SNHXR5KORBY7U/action/citation_signature","submit_replication":"https://pith.science/pith/5IEDEVPE6B6N6SNHXR5KORBY7U/action/replication_record"}},"created_at":"2026-07-05T03:26:17.573544+00:00","updated_at":"2026-07-05T03:26:17.573544+00:00"}