{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:E7ELTB7AFV7OE5TBX76ESBUMMF","short_pith_number":"pith:E7ELTB7A","schema_version":"1.0","canonical_sha256":"27c8b987e02d7ee27661bffc49068c615c5bdf88ec1d205041e49fd7553a7e8b","source":{"kind":"arxiv","id":"2502.14010","version":1},"attestation_state":"computed","paper":{"title":"Which Attention Heads Matter for In-Context Learning?","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Jacob Steinhardt, Kayo Yin","submitted_at":"2025-02-19T12:25:02Z","abstract_excerpt":"Large language models (LLMs) exhibit impressive in-context learning (ICL) capability, enabling them to perform new tasks using only a few demonstrations in the prompt. Two different mechanisms have been proposed to explain ICL: induction heads that find and copy relevant tokens, and function vector (FV) heads whose activations compute a latent encoding of the ICL task. To better understand which of the two distinct mechanisms drives ICL, we study and compare induction heads and FV heads in 12 language models.\n  Through detailed ablations, we discover that few-shot ICL performance depends prima"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.14010","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2025-02-19T12:25:02Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"6125ae2097a108715a210b0f464e05254e4ae26eeb69d7d0d8791bac27ef3589","abstract_canon_sha256":"9176b9fc56ba79166fb6e8ed20d3eacc58e8877236c1947c87bba388a439a8de"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:57:34.114533Z","signature_b64":"xtsbmOcG/sG9zjiNE1lI3bNMox3S5J02lxEBL7XjkL5kz8x2k29FInm9XNWiKKbqTu7km20k96DfN4xNwGnxDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"27c8b987e02d7ee27661bffc49068c615c5bdf88ec1d205041e49fd7553a7e8b","last_reissued_at":"2026-07-05T10:57:34.114031Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:57:34.114031Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Which Attention Heads Matter for In-Context Learning?","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Jacob Steinhardt, Kayo Yin","submitted_at":"2025-02-19T12:25:02Z","abstract_excerpt":"Large language models (LLMs) exhibit impressive in-context learning (ICL) capability, enabling them to perform new tasks using only a few demonstrations in the prompt. Two different mechanisms have been proposed to explain ICL: induction heads that find and copy relevant tokens, and function vector (FV) heads whose activations compute a latent encoding of the ICL task. To better understand which of the two distinct mechanisms drives ICL, we study and compare induction heads and FV heads in 12 language models.\n  Through detailed ablations, we discover that few-shot ICL performance depends prima"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.14010","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.14010/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.14010","created_at":"2026-07-05T10:57:34.114087+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.14010v1","created_at":"2026-07-05T10:57:34.114087+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.14010","created_at":"2026-07-05T10:57:34.114087+00:00"},{"alias_kind":"pith_short_12","alias_value":"E7ELTB7AFV7O","created_at":"2026-07-05T10:57:34.114087+00:00"},{"alias_kind":"pith_short_16","alias_value":"E7ELTB7AFV7OE5TB","created_at":"2026-07-05T10:57:34.114087+00:00"},{"alias_kind":"pith_short_8","alias_value":"E7ELTB7A","created_at":"2026-07-05T10:57:34.114087+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":9,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.24467","citing_title":"CompressKV: Semantic-Retrieval-Guided KV-Cache Compression for Resource-Efficient Long-Context LLM Inference","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2606.19317","citing_title":"Explaining Attention with Program Synthesis","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2606.19317","citing_title":"Explaining Attention with Program Synthesis","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2606.07604","citing_title":"Contribution Weights: A Geometrical Analysis of Self-Attention Transformers","ref_index":45,"is_internal_anchor":false},{"citing_arxiv_id":"2605.20730","citing_title":"Distributional Alignment as a Criterion for Designing Task Vectors in In-Context Learning","ref_index":51,"is_internal_anchor":false},{"citing_arxiv_id":"2509.24164","citing_title":"Localizing Task Recognition and Task Learning in In-Context Learning via Attention Head Analysis","ref_index":46,"is_internal_anchor":false},{"citing_arxiv_id":"2512.13751","citing_title":"MIDUS: Memory-Infused Depth Up-Scaling","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2604.23371","citing_title":"When Context Sticks: Studying Interference in In-Context Learning","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05115","citing_title":"Manifold Steering Reveals the Shared Geometry of Neural Network Representation and Behavior","ref_index":231,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/E7ELTB7AFV7OE5TBX76ESBUMMF","json":"https://pith.science/pith/E7ELTB7AFV7OE5TBX76ESBUMMF.json","graph_json":"https://pith.science/api/pith-number/E7ELTB7AFV7OE5TBX76ESBUMMF/graph.json","events_json":"https://pith.science/api/pith-number/E7ELTB7AFV7OE5TBX76ESBUMMF/events.json","paper":"https://pith.science/paper/E7ELTB7A"},"agent_actions":{"view_html":"https://pith.science/pith/E7ELTB7AFV7OE5TBX76ESBUMMF","download_json":"https://pith.science/pith/E7ELTB7AFV7OE5TBX76ESBUMMF.json","view_paper":"https://pith.science/paper/E7ELTB7A","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.14010&json=true","fetch_graph":"https://pith.science/api/pith-number/E7ELTB7AFV7OE5TBX76ESBUMMF/graph.json","fetch_events":"https://pith.science/api/pith-number/E7ELTB7AFV7OE5TBX76ESBUMMF/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/E7ELTB7AFV7OE5TBX76ESBUMMF/action/timestamp_anchor","attest_storage":"https://pith.science/pith/E7ELTB7AFV7OE5TBX76ESBUMMF/action/storage_attestation","attest_author":"https://pith.science/pith/E7ELTB7AFV7OE5TBX76ESBUMMF/action/author_attestation","sign_citation":"https://pith.science/pith/E7ELTB7AFV7OE5TBX76ESBUMMF/action/citation_signature","submit_replication":"https://pith.science/pith/E7ELTB7AFV7OE5TBX76ESBUMMF/action/replication_record"}},"created_at":"2026-07-05T10:57:34.114087+00:00","updated_at":"2026-07-05T10:57:34.114087+00:00"}