{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:V2P2W4ZMFQ76PIYTX6XNJIJEMW","short_pith_number":"pith:V2P2W4ZM","schema_version":"1.0","canonical_sha256":"ae9fab732c2c3fe7a313bfaed4a12465a90c8b4f5003ef1df8d9f7071decb30e","source":{"kind":"arxiv","id":"2402.02872","version":3},"attestation_state":"computed","paper":{"title":"How do Large Language Models Learn In-Context? Query and Key Matrices of In-Context Heads are Two Towers for Metric Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Sophia Ananiadou, Zeping Yu","submitted_at":"2024-02-05T10:39:32Z","abstract_excerpt":"We investigate the mechanism of in-context learning (ICL) on sentence classification tasks with semantically-unrelated labels (\"foo\"/\"bar\"). We find intervening in only 1\\% heads (named \"in-context heads\") significantly affects ICL accuracy from 87.6\\% to 24.4\\%. To understand this phenomenon, we analyze the value-output vectors in these heads and discover that the vectors at each label position contain substantial information about the corresponding labels. Furthermore, we observe that the prediction shift from \"foo\" to \"bar\" is due to the respective reduction and increase in these heads' att"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2402.02872","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-02-05T10:39:32Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"0b99ea2f561f10998814d442f36646d9caeaafc4fe0976a09797264e6627cb5e","abstract_canon_sha256":"6abcd80cf9fce7e95747b8e835acb6994bb1defa22123832d37a79e3f55bd616"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:11:24.322674Z","signature_b64":"Nz0fmKM9VsmmrAKIRET+zZr5SRiKIa/iX84ZJrlvuvrIBT3Uau19kd+AorgtzCnxhpWMggdUO+AJzJsRkrXhDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ae9fab732c2c3fe7a313bfaed4a12465a90c8b4f5003ef1df8d9f7071decb30e","last_reissued_at":"2026-07-05T09:11:24.322195Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:11:24.322195Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"How do Large Language Models Learn In-Context? Query and Key Matrices of In-Context Heads are Two Towers for Metric Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Sophia Ananiadou, Zeping Yu","submitted_at":"2024-02-05T10:39:32Z","abstract_excerpt":"We investigate the mechanism of in-context learning (ICL) on sentence classification tasks with semantically-unrelated labels (\"foo\"/\"bar\"). We find intervening in only 1\\% heads (named \"in-context heads\") significantly affects ICL accuracy from 87.6\\% to 24.4\\%. To understand this phenomenon, we analyze the value-output vectors in these heads and discover that the vectors at each label position contain substantial information about the corresponding labels. Furthermore, we observe that the prediction shift from \"foo\" to \"bar\" is due to the respective reduction and increase in these heads' att"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2402.02872","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2402.02872/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2402.02872","created_at":"2026-07-05T09:11:24.322252+00:00"},{"alias_kind":"arxiv_version","alias_value":"2402.02872v3","created_at":"2026-07-05T09:11:24.322252+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2402.02872","created_at":"2026-07-05T09:11:24.322252+00:00"},{"alias_kind":"pith_short_12","alias_value":"V2P2W4ZMFQ76","created_at":"2026-07-05T09:11:24.322252+00:00"},{"alias_kind":"pith_short_16","alias_value":"V2P2W4ZMFQ76PIYT","created_at":"2026-07-05T09:11:24.322252+00:00"},{"alias_kind":"pith_short_8","alias_value":"V2P2W4ZM","created_at":"2026-07-05T09:11:24.322252+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2509.24164","citing_title":"Localizing Task Recognition and Task Learning in In-Context Learning via Attention Head Analysis","ref_index":47,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/V2P2W4ZMFQ76PIYTX6XNJIJEMW","json":"https://pith.science/pith/V2P2W4ZMFQ76PIYTX6XNJIJEMW.json","graph_json":"https://pith.science/api/pith-number/V2P2W4ZMFQ76PIYTX6XNJIJEMW/graph.json","events_json":"https://pith.science/api/pith-number/V2P2W4ZMFQ76PIYTX6XNJIJEMW/events.json","paper":"https://pith.science/paper/V2P2W4ZM"},"agent_actions":{"view_html":"https://pith.science/pith/V2P2W4ZMFQ76PIYTX6XNJIJEMW","download_json":"https://pith.science/pith/V2P2W4ZMFQ76PIYTX6XNJIJEMW.json","view_paper":"https://pith.science/paper/V2P2W4ZM","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2402.02872&json=true","fetch_graph":"https://pith.science/api/pith-number/V2P2W4ZMFQ76PIYTX6XNJIJEMW/graph.json","fetch_events":"https://pith.science/api/pith-number/V2P2W4ZMFQ76PIYTX6XNJIJEMW/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/V2P2W4ZMFQ76PIYTX6XNJIJEMW/action/timestamp_anchor","attest_storage":"https://pith.science/pith/V2P2W4ZMFQ76PIYTX6XNJIJEMW/action/storage_attestation","attest_author":"https://pith.science/pith/V2P2W4ZMFQ76PIYTX6XNJIJEMW/action/author_attestation","sign_citation":"https://pith.science/pith/V2P2W4ZMFQ76PIYTX6XNJIJEMW/action/citation_signature","submit_replication":"https://pith.science/pith/V2P2W4ZMFQ76PIYTX6XNJIJEMW/action/replication_record"}},"created_at":"2026-07-05T09:11:24.322252+00:00","updated_at":"2026-07-05T09:11:24.322252+00:00"}