{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:R7KUQSP3X7L3VNSYAC7NPGB5X3","short_pith_number":"pith:R7KUQSP3","schema_version":"1.0","canonical_sha256":"8fd54849fbbfd7bab65800bed7983dbee1b98b9afce8dfbd05f7c56d6e703267","source":{"kind":"arxiv","id":"2502.21212","version":1},"attestation_state":"computed","paper":{"title":"Transformers Learn to Implement Multi-step Gradient Descent with Chain of Thought","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Jason D. Lee, Jianhao Huang, Zixuan Wang","submitted_at":"2025-02-28T16:40:38Z","abstract_excerpt":"Chain of Thought (CoT) prompting has been shown to significantly improve the performance of large language models (LLMs), particularly in arithmetic and reasoning tasks, by instructing the model to produce intermediate reasoning steps. Despite the remarkable empirical success of CoT and its theoretical advantages in enhancing expressivity, the mechanisms underlying CoT training remain largely unexplored. In this paper, we study the training dynamics of transformers over a CoT objective on an in-context weight prediction task for linear regression. We prove that while a one-layer linear transfo"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.21212","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2025-02-28T16:40:38Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"ad4adb0996e864286319e16f79b019652988432cbdf063c5a60497ce19a06f10","abstract_canon_sha256":"eff0ee39ad03c5432db0ae491039b332c2a335ceefdad577db0734c80bf68d03"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:21:46.291353Z","signature_b64":"wHMLkJH9wu1tPlXcrPQBJCxDe8TR1RT4GiYf2iavw2E1Xvg1gZAfC1VSqSCnxrn11bIRCKh3O7D2PCdR+i2EAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"8fd54849fbbfd7bab65800bed7983dbee1b98b9afce8dfbd05f7c56d6e703267","last_reissued_at":"2026-07-05T10:21:46.290865Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:21:46.290865Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Transformers Learn to Implement Multi-step Gradient Descent with Chain of Thought","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Jason D. Lee, Jianhao Huang, Zixuan Wang","submitted_at":"2025-02-28T16:40:38Z","abstract_excerpt":"Chain of Thought (CoT) prompting has been shown to significantly improve the performance of large language models (LLMs), particularly in arithmetic and reasoning tasks, by instructing the model to produce intermediate reasoning steps. Despite the remarkable empirical success of CoT and its theoretical advantages in enhancing expressivity, the mechanisms underlying CoT training remain largely unexplored. In this paper, we study the training dynamics of transformers over a CoT objective on an in-context weight prediction task for linear regression. We prove that while a one-layer linear transfo"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.21212","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.21212/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.21212","created_at":"2026-07-05T10:21:46.290922+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.21212v1","created_at":"2026-07-05T10:21:46.290922+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.21212","created_at":"2026-07-05T10:21:46.290922+00:00"},{"alias_kind":"pith_short_12","alias_value":"R7KUQSP3X7L3","created_at":"2026-07-05T10:21:46.290922+00:00"},{"alias_kind":"pith_short_16","alias_value":"R7KUQSP3X7L3VNSY","created_at":"2026-07-05T10:21:46.290922+00:00"},{"alias_kind":"pith_short_8","alias_value":"R7KUQSP3","created_at":"2026-07-05T10:21:46.290922+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.28600","citing_title":"Transformers Provably Learn to Internalize Chain-of-Thought","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2510.25741","citing_title":"Scaling Latent Reasoning via Looped Language Models","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2604.22951","citing_title":"The Power of Power Law: Asymmetry Enables Compositional Reasoning","ref_index":23,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/R7KUQSP3X7L3VNSYAC7NPGB5X3","json":"https://pith.science/pith/R7KUQSP3X7L3VNSYAC7NPGB5X3.json","graph_json":"https://pith.science/api/pith-number/R7KUQSP3X7L3VNSYAC7NPGB5X3/graph.json","events_json":"https://pith.science/api/pith-number/R7KUQSP3X7L3VNSYAC7NPGB5X3/events.json","paper":"https://pith.science/paper/R7KUQSP3"},"agent_actions":{"view_html":"https://pith.science/pith/R7KUQSP3X7L3VNSYAC7NPGB5X3","download_json":"https://pith.science/pith/R7KUQSP3X7L3VNSYAC7NPGB5X3.json","view_paper":"https://pith.science/paper/R7KUQSP3","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.21212&json=true","fetch_graph":"https://pith.science/api/pith-number/R7KUQSP3X7L3VNSYAC7NPGB5X3/graph.json","fetch_events":"https://pith.science/api/pith-number/R7KUQSP3X7L3VNSYAC7NPGB5X3/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/R7KUQSP3X7L3VNSYAC7NPGB5X3/action/timestamp_anchor","attest_storage":"https://pith.science/pith/R7KUQSP3X7L3VNSYAC7NPGB5X3/action/storage_attestation","attest_author":"https://pith.science/pith/R7KUQSP3X7L3VNSYAC7NPGB5X3/action/author_attestation","sign_citation":"https://pith.science/pith/R7KUQSP3X7L3VNSYAC7NPGB5X3/action/citation_signature","submit_replication":"https://pith.science/pith/R7KUQSP3X7L3VNSYAC7NPGB5X3/action/replication_record"}},"created_at":"2026-07-05T10:21:46.290922+00:00","updated_at":"2026-07-05T10:21:46.290922+00:00"}