{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:VLFIUOWSE77VAJ3FYM2IDAISE4","short_pith_number":"pith:VLFIUOWS","schema_version":"1.0","canonical_sha256":"aaca8a3ad227ff502765c3348181122736e684ffcd7b727ea118fa8072d6960d","source":{"kind":"arxiv","id":"2501.16265","version":2},"attestation_state":"computed","paper":{"title":"Training Dynamics of In-Context Learning in Linear Attention","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Aaditya K. Singh, Andrew Saxe, Peter E. Latham, Yedi Zhang","submitted_at":"2025-01-27T18:03:00Z","abstract_excerpt":"While attention-based models have demonstrated the remarkable ability of in-context learning (ICL), the theoretical understanding of how these models acquired this ability through gradient descent training is still preliminary. Towards answering this question, we study the gradient descent dynamics of multi-head linear self-attention trained for in-context linear regression. We examine two parametrizations of linear self-attention: one with the key and query weights merged as a single matrix (common in theoretical studies), and one with separate key and query matrices (closer to practical sett"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2501.16265","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.LG","submitted_at":"2025-01-27T18:03:00Z","cross_cats_sorted":[],"title_canon_sha256":"6d5f37db3377250d5c1c4a5c5f305f68d42ea3bb5a2723bb6ef8d1f6dbde5b6f","abstract_canon_sha256":"1605d0dca51f2a559fa6d9d99198818a8454af14e9a33ffe1579c01a45c73a60"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:10:50.773429Z","signature_b64":"SpsvEcP6g5N0E1fjX9H/bHnVOcG3chgquI9rpTJJM++dXRMBuQxR4kUbGd/05stkwoFF2pI6T6u9VSEbmW4nCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"aaca8a3ad227ff502765c3348181122736e684ffcd7b727ea118fa8072d6960d","last_reissued_at":"2026-07-05T11:10:50.772927Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:10:50.772927Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Training Dynamics of In-Context Learning in Linear Attention","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Aaditya K. Singh, Andrew Saxe, Peter E. Latham, Yedi Zhang","submitted_at":"2025-01-27T18:03:00Z","abstract_excerpt":"While attention-based models have demonstrated the remarkable ability of in-context learning (ICL), the theoretical understanding of how these models acquired this ability through gradient descent training is still preliminary. Towards answering this question, we study the gradient descent dynamics of multi-head linear self-attention trained for in-context linear regression. We examine two parametrizations of linear self-attention: one with the key and query weights merged as a single matrix (common in theoretical studies), and one with separate key and query matrices (closer to practical sett"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2501.16265","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2501.16265/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2501.16265","created_at":"2026-07-05T11:10:50.772993+00:00"},{"alias_kind":"arxiv_version","alias_value":"2501.16265v2","created_at":"2026-07-05T11:10:50.772993+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2501.16265","created_at":"2026-07-05T11:10:50.772993+00:00"},{"alias_kind":"pith_short_12","alias_value":"VLFIUOWSE77V","created_at":"2026-07-05T11:10:50.772993+00:00"},{"alias_kind":"pith_short_16","alias_value":"VLFIUOWSE77VAJ3F","created_at":"2026-07-05T11:10:50.772993+00:00"},{"alias_kind":"pith_short_8","alias_value":"VLFIUOWS","created_at":"2026-07-05T11:10:50.772993+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":7,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.09283","citing_title":"Towards personalised intervention: A causal-dynamical framework to determine psychological treatment trajectories","ref_index":132,"is_internal_anchor":false},{"citing_arxiv_id":"2606.03217","citing_title":"An Asymptotic Theory of Chain-of-Thought in In-Context Learning","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2606.05219","citing_title":"Gradient Descent with Large Step Size Restores Symmetry in Deep Linear Networks with Multi-Pathway","ref_index":41,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08475","citing_title":"Transformers Can Implement Preconditioned Richardson Iteration for In-Context Gaussian Kernel Regression","ref_index":44,"is_internal_anchor":false},{"citing_arxiv_id":"2603.01097","citing_title":"Understanding LoRA as Knowledge Memory: An Empirical Analysis","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08475","citing_title":"Transformers Can Implement Preconditioned Richardson Iteration for In-Context Gaussian Kernel Regression","ref_index":44,"is_internal_anchor":false},{"citing_arxiv_id":"2604.10946","citing_title":"Learning to Adapt: In-Context Learning Beyond Stationarity","ref_index":55,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/VLFIUOWSE77VAJ3FYM2IDAISE4","json":"https://pith.science/pith/VLFIUOWSE77VAJ3FYM2IDAISE4.json","graph_json":"https://pith.science/api/pith-number/VLFIUOWSE77VAJ3FYM2IDAISE4/graph.json","events_json":"https://pith.science/api/pith-number/VLFIUOWSE77VAJ3FYM2IDAISE4/events.json","paper":"https://pith.science/paper/VLFIUOWS"},"agent_actions":{"view_html":"https://pith.science/pith/VLFIUOWSE77VAJ3FYM2IDAISE4","download_json":"https://pith.science/pith/VLFIUOWSE77VAJ3FYM2IDAISE4.json","view_paper":"https://pith.science/paper/VLFIUOWS","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2501.16265&json=true","fetch_graph":"https://pith.science/api/pith-number/VLFIUOWSE77VAJ3FYM2IDAISE4/graph.json","fetch_events":"https://pith.science/api/pith-number/VLFIUOWSE77VAJ3FYM2IDAISE4/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/VLFIUOWSE77VAJ3FYM2IDAISE4/action/timestamp_anchor","attest_storage":"https://pith.science/pith/VLFIUOWSE77VAJ3FYM2IDAISE4/action/storage_attestation","attest_author":"https://pith.science/pith/VLFIUOWSE77VAJ3FYM2IDAISE4/action/author_attestation","sign_citation":"https://pith.science/pith/VLFIUOWSE77VAJ3FYM2IDAISE4/action/citation_signature","submit_replication":"https://pith.science/pith/VLFIUOWSE77VAJ3FYM2IDAISE4/action/replication_record"}},"created_at":"2026-07-05T11:10:50.772993+00:00","updated_at":"2026-07-05T11:10:50.772993+00:00"}