{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:5EI4AKGLB5W43RIBFIMBCYT4YI","short_pith_number":"pith:5EI4AKGL","schema_version":"1.0","canonical_sha256":"e911c028cb0f6dcdc5012a1811627cc225c6dce8db830d4a06414e33efadaf5c","source":{"kind":"arxiv","id":"2304.14997","version":4},"attestation_state":"computed","paper":{"title":"Towards Automated Circuit Discovery for Mechanistic Interpretability","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Adri\\`a Garriga-Alonso, Aengus Lynch, Arthur Conmy, Augustine N. Mavor-Parker, Stefan Heimersheim","submitted_at":"2023-04-28T17:36:53Z","abstract_excerpt":"Through considerable effort and intuition, several recent works have reverse-engineered nontrivial behaviors of transformer models. This paper systematizes the mechanistic interpretability process they followed. First, researchers choose a metric and dataset that elicit the desired model behavior. Then, they apply activation patching to find which abstract neural network units are involved in the behavior. By varying the dataset, metric, and units under investigation, researchers can understand the functionality of each component. We automate one of the process' steps: to identify the circuit "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2304.14997","kind":"arxiv","version":4},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2023-04-28T17:36:53Z","cross_cats_sorted":[],"title_canon_sha256":"967cdeb4ffeedab758609f81f8eeada1617f415252dbe3d6918ca893ccb723d9","abstract_canon_sha256":"44381c8c5d968c594073dcb6cc52e4e36874fa8d8b33d717b5b290392efaa115"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:06:26.203280Z","signature_b64":"XuFZ6ztGndwYE/y5tR7TqdgbgVb5tsk5mdZp9Vq0W1cY1eYKvHAeJEQfJMN6m0oXCZriYj3HhVMHus1fuHCOAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"e911c028cb0f6dcdc5012a1811627cc225c6dce8db830d4a06414e33efadaf5c","last_reissued_at":"2026-07-05T07:06:26.202768Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:06:26.202768Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Towards Automated Circuit Discovery for Mechanistic Interpretability","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Adri\\`a Garriga-Alonso, Aengus Lynch, Arthur Conmy, Augustine N. Mavor-Parker, Stefan Heimersheim","submitted_at":"2023-04-28T17:36:53Z","abstract_excerpt":"Through considerable effort and intuition, several recent works have reverse-engineered nontrivial behaviors of transformer models. This paper systematizes the mechanistic interpretability process they followed. First, researchers choose a metric and dataset that elicit the desired model behavior. Then, they apply activation patching to find which abstract neural network units are involved in the behavior. By varying the dataset, metric, and units under investigation, researchers can understand the functionality of each component. We automate one of the process' steps: to identify the circuit "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2304.14997","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2304.14997/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2304.14997","created_at":"2026-07-05T07:06:26.202840+00:00"},{"alias_kind":"arxiv_version","alias_value":"2304.14997v4","created_at":"2026-07-05T07:06:26.202840+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2304.14997","created_at":"2026-07-05T07:06:26.202840+00:00"},{"alias_kind":"pith_short_12","alias_value":"5EI4AKGLB5W4","created_at":"2026-07-05T07:06:26.202840+00:00"},{"alias_kind":"pith_short_16","alias_value":"5EI4AKGLB5W43RIB","created_at":"2026-07-05T07:06:26.202840+00:00"},{"alias_kind":"pith_short_8","alias_value":"5EI4AKGL","created_at":"2026-07-05T07:06:26.202840+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":18,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.07316","citing_title":"Mechanistic Interpretability for Neural Networks: Circuits, Sparse Features and Symbolic Reasoning","ref_index":20,"is_internal_anchor":true},{"citing_arxiv_id":"2606.19317","citing_title":"Explaining Attention with Program Synthesis","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2607.00924","citing_title":"Graph-Native Reinforcement Learning Enables Traceable Scientific Hypothesis Generation through Conceptual Recombination","ref_index":58,"is_internal_anchor":false},{"citing_arxiv_id":"2606.19317","citing_title":"Explaining Attention with Program Synthesis","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2309.08600","citing_title":"Sparse Autoencoders Find Highly Interpretable Features in Language Models","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2605.22462","citing_title":"From Correlation to Cause: A Five-Stage Methodology for Feature Analysis in Transformer Language Models","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12991","citing_title":"Not Just RLHF: Why Alignment Alone Won't Fix Multi-Agent Sycophancy","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2404.15255","citing_title":"How to use and interpret activation patching","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2408.05147","citing_title":"Gemma Scope: Open Sparse Autoencoders Everywhere All At Once on Gemma 2","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2605.15183","citing_title":"When Are Two Networks the Same? Tensor Similarity for Mechanistic Interpretability","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12809","citing_title":"Correcting Influence: Unboxing LLM Outputs with Orthogonal Latent Spaces","ref_index":77,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12991","citing_title":"Not Just RLHF: Why Alignment Alone Won't Fix Multi-Agent Sycophancy","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2605.13625","citing_title":"How to Interpret Agent Behavior","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2604.03045","citing_title":"STEAR: Layer-Aware Spatiotemporal Evidence Intervention for Hallucination Mitigation in Video Large Language Models","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09881","citing_title":"Dissecting Jet-Tagger Through Mechanistic Interpretability","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2604.19826","citing_title":"Co-Located Tests, Better AI Code: How Test Syntax Structure Affects Foundation Model Code Generation","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2604.22128","citing_title":"Dissociating Decodability and Causal Use in Bracket-Sequence Transformers","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06335","citing_title":"Eliciting associations between clinical variables from LLMs via comparison questions across populations","ref_index":6,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/5EI4AKGLB5W43RIBFIMBCYT4YI","json":"https://pith.science/pith/5EI4AKGLB5W43RIBFIMBCYT4YI.json","graph_json":"https://pith.science/api/pith-number/5EI4AKGLB5W43RIBFIMBCYT4YI/graph.json","events_json":"https://pith.science/api/pith-number/5EI4AKGLB5W43RIBFIMBCYT4YI/events.json","paper":"https://pith.science/paper/5EI4AKGL"},"agent_actions":{"view_html":"https://pith.science/pith/5EI4AKGLB5W43RIBFIMBCYT4YI","download_json":"https://pith.science/pith/5EI4AKGLB5W43RIBFIMBCYT4YI.json","view_paper":"https://pith.science/paper/5EI4AKGL","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2304.14997&json=true","fetch_graph":"https://pith.science/api/pith-number/5EI4AKGLB5W43RIBFIMBCYT4YI/graph.json","fetch_events":"https://pith.science/api/pith-number/5EI4AKGLB5W43RIBFIMBCYT4YI/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/5EI4AKGLB5W43RIBFIMBCYT4YI/action/timestamp_anchor","attest_storage":"https://pith.science/pith/5EI4AKGLB5W43RIBFIMBCYT4YI/action/storage_attestation","attest_author":"https://pith.science/pith/5EI4AKGLB5W43RIBFIMBCYT4YI/action/author_attestation","sign_citation":"https://pith.science/pith/5EI4AKGLB5W43RIBFIMBCYT4YI/action/citation_signature","submit_replication":"https://pith.science/pith/5EI4AKGLB5W43RIBFIMBCYT4YI/action/replication_record"}},"created_at":"2026-07-05T07:06:26.202840+00:00","updated_at":"2026-07-05T07:06:26.202840+00:00"}