{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:6CJWDV566PEGUO7KCQNNKGVSJ7","short_pith_number":"pith:6CJWDV56","schema_version":"1.0","canonical_sha256":"f09361d7bef3c86a3bea141ad51ab24ff519717d527cb5e54970f51fa348ba2b","source":{"kind":"arxiv","id":"2310.04625","version":1},"attestation_state":"computed","paper":{"title":"Copy Suppression: Comprehensively Understanding an Attention Head","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Arthur Conmy, Callum McDougall, Cody Rushing, Neel Nanda, Thomas McGrath","submitted_at":"2023-10-06T23:37:24Z","abstract_excerpt":"We present a single attention head in GPT-2 Small that has one main role across the entire training distribution. If components in earlier layers predict a certain token, and this token appears earlier in the context, the head suppresses it: we call this copy suppression. Attention Head 10.7 (L10H7) suppresses naive copying behavior which improves overall model calibration. This explains why multiple prior works studying certain narrow tasks found negative heads that systematically favored the wrong answer. We uncover the mechanism that the Negative Heads use for copy suppression with weights-"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2310.04625","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2023-10-06T23:37:24Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"ba01c30432dfc13481f3deb393571df30be3696928656834d31450ebbc267acb","abstract_canon_sha256":"15ea11c590581ea5f5b8eecad5c3fa3fa9a2e181d19572e8409bbc50cda1784e"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:58:20.843699Z","signature_b64":"2YwyLSYqFAOv91td92M6FtOzRpYAlrqSE5gAxXS/zwq9eYNSxN5xrBDBv4vhJa4COYYxOCBoJlaf5xksfn9CBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f09361d7bef3c86a3bea141ad51ab24ff519717d527cb5e54970f51fa348ba2b","last_reissued_at":"2026-07-05T06:58:20.843286Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:58:20.843286Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Copy Suppression: Comprehensively Understanding an Attention Head","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Arthur Conmy, Callum McDougall, Cody Rushing, Neel Nanda, Thomas McGrath","submitted_at":"2023-10-06T23:37:24Z","abstract_excerpt":"We present a single attention head in GPT-2 Small that has one main role across the entire training distribution. If components in earlier layers predict a certain token, and this token appears earlier in the context, the head suppresses it: we call this copy suppression. Attention Head 10.7 (L10H7) suppresses naive copying behavior which improves overall model calibration. This explains why multiple prior works studying certain narrow tasks found negative heads that systematically favored the wrong answer. We uncover the mechanism that the Negative Heads use for copy suppression with weights-"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2310.04625","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2310.04625/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2310.04625","created_at":"2026-07-05T06:58:20.843356+00:00"},{"alias_kind":"arxiv_version","alias_value":"2310.04625v1","created_at":"2026-07-05T06:58:20.843356+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2310.04625","created_at":"2026-07-05T06:58:20.843356+00:00"},{"alias_kind":"pith_short_12","alias_value":"6CJWDV566PEG","created_at":"2026-07-05T06:58:20.843356+00:00"},{"alias_kind":"pith_short_16","alias_value":"6CJWDV566PEGUO7K","created_at":"2026-07-05T06:58:20.843356+00:00"},{"alias_kind":"pith_short_8","alias_value":"6CJWDV56","created_at":"2026-07-05T06:58:20.843356+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":7,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.08292","citing_title":"Necessary, Decodable and Reversible, Yet Not Transferable: A Stress Test for Attention-Head Role Claims","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2606.28153","citing_title":"Robust Harmful Features Under Jailbreak Attacks: Mechanistic Evidence from Attention Head Specialization in Large Language Models","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2605.22870","citing_title":"The Readout Shortcut: Positional Number Copying Dominates Arithmetic CoT Readout in Small Language Models","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2404.15255","citing_title":"How to use and interpret activation patching","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12809","citing_title":"Correcting Influence: Unboxing LLM Outputs with Orthogonal Latent Spaces","ref_index":99,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09314","citing_title":"How LLMs Are Persuaded: A Few Attention Heads, Rerouted","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06510","citing_title":"Is One Layer Enough? Understanding Inference Dynamics in Tabular Foundation Models","ref_index":31,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/6CJWDV566PEGUO7KCQNNKGVSJ7","json":"https://pith.science/pith/6CJWDV566PEGUO7KCQNNKGVSJ7.json","graph_json":"https://pith.science/api/pith-number/6CJWDV566PEGUO7KCQNNKGVSJ7/graph.json","events_json":"https://pith.science/api/pith-number/6CJWDV566PEGUO7KCQNNKGVSJ7/events.json","paper":"https://pith.science/paper/6CJWDV56"},"agent_actions":{"view_html":"https://pith.science/pith/6CJWDV566PEGUO7KCQNNKGVSJ7","download_json":"https://pith.science/pith/6CJWDV566PEGUO7KCQNNKGVSJ7.json","view_paper":"https://pith.science/paper/6CJWDV56","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2310.04625&json=true","fetch_graph":"https://pith.science/api/pith-number/6CJWDV566PEGUO7KCQNNKGVSJ7/graph.json","fetch_events":"https://pith.science/api/pith-number/6CJWDV566PEGUO7KCQNNKGVSJ7/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/6CJWDV566PEGUO7KCQNNKGVSJ7/action/timestamp_anchor","attest_storage":"https://pith.science/pith/6CJWDV566PEGUO7KCQNNKGVSJ7/action/storage_attestation","attest_author":"https://pith.science/pith/6CJWDV566PEGUO7KCQNNKGVSJ7/action/author_attestation","sign_citation":"https://pith.science/pith/6CJWDV566PEGUO7KCQNNKGVSJ7/action/citation_signature","submit_replication":"https://pith.science/pith/6CJWDV566PEGUO7KCQNNKGVSJ7/action/replication_record"}},"created_at":"2026-07-05T06:58:20.843356+00:00","updated_at":"2026-07-05T06:58:20.843356+00:00"}