{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:WFY4HT7U3WZOD3BWDX4GAKADWZ","short_pith_number":"pith:WFY4HT7U","schema_version":"1.0","canonical_sha256":"b171c3cff4ddb2e1ec361df8602803b65e67837a08e930c68e23311d351f5015","source":{"kind":"arxiv","id":"2306.00297","version":2},"attestation_state":"computed","paper":{"title":"Transformers learn to implement preconditioned gradient descent for in-context learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Hadi Daneshmand, Kwangjun Ahn, Suvrit Sra, Xiang Cheng","submitted_at":"2023-06-01T02:35:57Z","abstract_excerpt":"Several recent works demonstrate that transformers can implement algorithms like gradient descent. By a careful construction of weights, these works show that multiple layers of transformers are expressive enough to simulate iterations of gradient descent. Going beyond the question of expressivity, we ask: Can transformers learn to implement such algorithms by training over random problem instances? To our knowledge, we make the first theoretical progress on this question via an analysis of the loss landscape for linear transformers trained over random instances of linear regression. For a sin"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2306.00297","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2023-06-01T02:35:57Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"b0aa54109330e45961397138f2e069bb219e0d85ad23a681db88e3dd525c00b1","abstract_canon_sha256":"9798e2c88b1f5770216d0f44b7bcc33a14c486d468d36df77699cb10c95a1cfc"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:11:15.949650Z","signature_b64":"IX2Hy4lk1I8zYJ0FenUrTYNBC85fWUYBHFWk1Lh3hR4e1gSH102UMxsyZTrOqI0vF5ywXN2axxBJXwCgdfzFDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b171c3cff4ddb2e1ec361df8602803b65e67837a08e930c68e23311d351f5015","last_reissued_at":"2026-07-05T07:11:15.949147Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:11:15.949147Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Transformers learn to implement preconditioned gradient descent for in-context learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Hadi Daneshmand, Kwangjun Ahn, Suvrit Sra, Xiang Cheng","submitted_at":"2023-06-01T02:35:57Z","abstract_excerpt":"Several recent works demonstrate that transformers can implement algorithms like gradient descent. By a careful construction of weights, these works show that multiple layers of transformers are expressive enough to simulate iterations of gradient descent. Going beyond the question of expressivity, we ask: Can transformers learn to implement such algorithms by training over random problem instances? To our knowledge, we make the first theoretical progress on this question via an analysis of the loss landscape for linear transformers trained over random instances of linear regression. For a sin"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2306.00297","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2306.00297/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2306.00297","created_at":"2026-07-05T07:11:15.949205+00:00"},{"alias_kind":"arxiv_version","alias_value":"2306.00297v2","created_at":"2026-07-05T07:11:15.949205+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2306.00297","created_at":"2026-07-05T07:11:15.949205+00:00"},{"alias_kind":"pith_short_12","alias_value":"WFY4HT7U3WZO","created_at":"2026-07-05T07:11:15.949205+00:00"},{"alias_kind":"pith_short_16","alias_value":"WFY4HT7U3WZOD3BW","created_at":"2026-07-05T07:11:15.949205+00:00"},{"alias_kind":"pith_short_8","alias_value":"WFY4HT7U","created_at":"2026-07-05T07:11:15.949205+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.11219","citing_title":"Afrispeech Semantics: Evaluating Audio Semantic Reasoning in Spoken Language Models Across Domains and Accents","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2606.07623","citing_title":"Finite Certificates for In-Context Determinacy and a Threshold Theory of Emergence in Language Models","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2502.05564","citing_title":"TabICL: A Tabular Foundation Model for In-Context Learning on Large Data","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09727","citing_title":"One for All: A Non-Linear Transformer can Enable Cross-Domain Generalization for In-Context Reinforcement Learning","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09498","citing_title":"Spectral Transformer Neural Processes","ref_index":1,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/WFY4HT7U3WZOD3BWDX4GAKADWZ","json":"https://pith.science/pith/WFY4HT7U3WZOD3BWDX4GAKADWZ.json","graph_json":"https://pith.science/api/pith-number/WFY4HT7U3WZOD3BWDX4GAKADWZ/graph.json","events_json":"https://pith.science/api/pith-number/WFY4HT7U3WZOD3BWDX4GAKADWZ/events.json","paper":"https://pith.science/paper/WFY4HT7U"},"agent_actions":{"view_html":"https://pith.science/pith/WFY4HT7U3WZOD3BWDX4GAKADWZ","download_json":"https://pith.science/pith/WFY4HT7U3WZOD3BWDX4GAKADWZ.json","view_paper":"https://pith.science/paper/WFY4HT7U","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2306.00297&json=true","fetch_graph":"https://pith.science/api/pith-number/WFY4HT7U3WZOD3BWDX4GAKADWZ/graph.json","fetch_events":"https://pith.science/api/pith-number/WFY4HT7U3WZOD3BWDX4GAKADWZ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/WFY4HT7U3WZOD3BWDX4GAKADWZ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/WFY4HT7U3WZOD3BWDX4GAKADWZ/action/storage_attestation","attest_author":"https://pith.science/pith/WFY4HT7U3WZOD3BWDX4GAKADWZ/action/author_attestation","sign_citation":"https://pith.science/pith/WFY4HT7U3WZOD3BWDX4GAKADWZ/action/citation_signature","submit_replication":"https://pith.science/pith/WFY4HT7U3WZOD3BWDX4GAKADWZ/action/replication_record"}},"created_at":"2026-07-05T07:11:15.949205+00:00","updated_at":"2026-07-05T07:11:15.949205+00:00"}