{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:IWEM3X4DOHFO6YLX4TLWVUHDUK","short_pith_number":"pith:IWEM3X4D","schema_version":"1.0","canonical_sha256":"4588cddf8371caef6177e4d76ad0e3a2950e23e55ccf2ef15be3be54117e1a85","source":{"kind":"arxiv","id":"2310.10348","version":2},"attestation_state":"computed","paper":{"title":"Attribution Patching Outperforms Automated Circuit Discovery","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Aaquib Syed, Arthur Conmy, Can Rager","submitted_at":"2023-10-16T12:34:43Z","abstract_excerpt":"Automated interpretability research has recently attracted attention as a potential research direction that could scale explanations of neural network behavior to large models. Existing automated circuit discovery work applies activation patching to identify subnetworks responsible for solving specific tasks (circuits). In this work, we show that a simple method based on attribution patching outperforms all existing methods while requiring just two forward passes and a backward pass. We apply a linear approximation to activation patching to estimate the importance of each edge in the computati"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2310.10348","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2023-10-16T12:34:43Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"d1f2d6e203f283afc40d6f6a497ec3a3dffd9240f6d419d1c5485f19f7a00840","abstract_canon_sha256":"945975e630565995035370c2390446b1e485ff50158e9e15feb3a1887a58a99d"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:14:18.988498Z","signature_b64":"n2OQgcdKCBI/EDQMxq0stabvV3orNcqH0Vc8L/D7eZAjEua8zdxmjq/Rq2otRCh5KG/Mx299VsnA50mRAIeEDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"4588cddf8371caef6177e4d76ad0e3a2950e23e55ccf2ef15be3be54117e1a85","last_reissued_at":"2026-07-05T07:14:18.987941Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:14:18.987941Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Attribution Patching Outperforms Automated Circuit Discovery","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Aaquib Syed, Arthur Conmy, Can Rager","submitted_at":"2023-10-16T12:34:43Z","abstract_excerpt":"Automated interpretability research has recently attracted attention as a potential research direction that could scale explanations of neural network behavior to large models. Existing automated circuit discovery work applies activation patching to identify subnetworks responsible for solving specific tasks (circuits). In this work, we show that a simple method based on attribution patching outperforms all existing methods while requiring just two forward passes and a backward pass. We apply a linear approximation to activation patching to estimate the importance of each edge in the computati"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2310.10348","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2310.10348/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2310.10348","created_at":"2026-07-05T07:14:18.987999+00:00"},{"alias_kind":"arxiv_version","alias_value":"2310.10348v2","created_at":"2026-07-05T07:14:18.987999+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2310.10348","created_at":"2026-07-05T07:14:18.987999+00:00"},{"alias_kind":"pith_short_12","alias_value":"IWEM3X4DOHFO","created_at":"2026-07-05T07:14:18.987999+00:00"},{"alias_kind":"pith_short_16","alias_value":"IWEM3X4DOHFO6YLX","created_at":"2026-07-05T07:14:18.987999+00:00"},{"alias_kind":"pith_short_8","alias_value":"IWEM3X4D","created_at":"2026-07-05T07:14:18.987999+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":16,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.05355","citing_title":"Faithfulness to Refusal: A Causal Audit of Neuron Selectors","ref_index":16,"is_internal_anchor":true},{"citing_arxiv_id":"2607.00267","citing_title":"Validating Causal Abstraction Metrics on Simulated Complex Systems","ref_index":165,"is_internal_anchor":false},{"citing_arxiv_id":"2606.29522","citing_title":"Do Models Read What They Write? Causal Registers in Scratchpad Reasoning","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2605.29634","citing_title":"Relational Rank Geometry in Transformers: Detecting and Steering Hidden-State Relation Frames","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2605.29358","citing_title":"Scaling Monosemanticity: Extracting Interpretable Features from Claude 3 Sonnet","ref_index":67,"is_internal_anchor":false},{"citing_arxiv_id":"2606.23591","citing_title":"Quantifying the Agreement Between Data-Influence and Data-Similarity to Understand LLM Behavior","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2605.20784","citing_title":"Interaction Locality in Hierarchical Recursive Reasoning","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2506.13727","citing_title":"Attribution-Guided Pruning for Insight and Control: Circuit Discovery and Targeted Correction in Small-scale LLMs","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12770","citing_title":"WriteSAE: Sparse Autoencoders for Recurrent State","ref_index":44,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11206","citing_title":"Instructions Shape Production of Language, not Processing","ref_index":92,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12770","citing_title":"WriteSAE: Sparse Autoencoders for Recurrent State","ref_index":44,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12809","citing_title":"Correcting Influence: Unboxing LLM Outputs with Orthogonal Latent Spaces","ref_index":95,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11206","citing_title":"Instructions Shape Production of Language, not Processing","ref_index":92,"is_internal_anchor":false},{"citing_arxiv_id":"2604.08846","citing_title":"Dictionary-Aligned Concept Control for Safeguarding Multimodal LLMs","ref_index":93,"is_internal_anchor":false},{"citing_arxiv_id":"2604.08192","citing_title":"Inside-Out: Measuring Generalization in Vision Transformers Through Inner Workings","ref_index":65,"is_internal_anchor":false},{"citing_arxiv_id":"2604.15780","citing_title":"Pruning Unsafe Tickets: A Resource-Efficient Framework for Safer and More Robust LLMs","ref_index":43,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/IWEM3X4DOHFO6YLX4TLWVUHDUK","json":"https://pith.science/pith/IWEM3X4DOHFO6YLX4TLWVUHDUK.json","graph_json":"https://pith.science/api/pith-number/IWEM3X4DOHFO6YLX4TLWVUHDUK/graph.json","events_json":"https://pith.science/api/pith-number/IWEM3X4DOHFO6YLX4TLWVUHDUK/events.json","paper":"https://pith.science/paper/IWEM3X4D"},"agent_actions":{"view_html":"https://pith.science/pith/IWEM3X4DOHFO6YLX4TLWVUHDUK","download_json":"https://pith.science/pith/IWEM3X4DOHFO6YLX4TLWVUHDUK.json","view_paper":"https://pith.science/paper/IWEM3X4D","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2310.10348&json=true","fetch_graph":"https://pith.science/api/pith-number/IWEM3X4DOHFO6YLX4TLWVUHDUK/graph.json","fetch_events":"https://pith.science/api/pith-number/IWEM3X4DOHFO6YLX4TLWVUHDUK/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/IWEM3X4DOHFO6YLX4TLWVUHDUK/action/timestamp_anchor","attest_storage":"https://pith.science/pith/IWEM3X4DOHFO6YLX4TLWVUHDUK/action/storage_attestation","attest_author":"https://pith.science/pith/IWEM3X4DOHFO6YLX4TLWVUHDUK/action/author_attestation","sign_citation":"https://pith.science/pith/IWEM3X4DOHFO6YLX4TLWVUHDUK/action/citation_signature","submit_replication":"https://pith.science/pith/IWEM3X4DOHFO6YLX4TLWVUHDUK/action/replication_record"}},"created_at":"2026-07-05T07:14:18.987999+00:00","updated_at":"2026-07-05T07:14:18.987999+00:00"}