{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:CTQW7ROD4YLKO5J77TJWVWGMZX","short_pith_number":"pith:CTQW7ROD","schema_version":"1.0","canonical_sha256":"14e16fc5c3e616a7753ffcd36ad8cccdd62fdb7408b150e23521e1cacf11535e","source":{"kind":"arxiv","id":"2305.16380","version":4},"attestation_state":"computed","paper":{"title":"Scan and Snap: Understanding Training Dynamics and Token Composition in 1-layer Transformer","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Beidi Chen, Simon Du, Yiping Wang, Yuandong Tian","submitted_at":"2023-05-25T15:59:13Z","abstract_excerpt":"Transformer architecture has shown impressive performance in multiple research domains and has become the backbone of many neural network models. However, there is limited understanding on how it works. In particular, with a simple predictive loss, how the representation emerges from the gradient \\emph{training dynamics} remains a mystery. In this paper, for 1-layer transformer with one self-attention layer plus one decoder layer, we analyze its SGD training dynamics for the task of next token prediction in a mathematically rigorous manner. We open the black box of the dynamic process of how t"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2305.16380","kind":"arxiv","version":4},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2023-05-25T15:59:13Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"da3305c2a42283a94449a369aea9a8f14b44df6a0cdc46b33fabe48d02d0161b","abstract_canon_sha256":"c2cba09150d78e929c4313e4489672c9637d13e1b6d552fe2db222c571556501"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:06:27.996077Z","signature_b64":"Fav0sPaERZnUqNuWjZwlOLvo7BzXN7j0pg6awIZuJDeKNELGUnSSzip3R8UmD5SDU/b+NXb/j9te38A0To2HBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"14e16fc5c3e616a7753ffcd36ad8cccdd62fdb7408b150e23521e1cacf11535e","last_reissued_at":"2026-07-05T07:06:27.995605Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:06:27.995605Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Scan and Snap: Understanding Training Dynamics and Token Composition in 1-layer Transformer","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Beidi Chen, Simon Du, Yiping Wang, Yuandong Tian","submitted_at":"2023-05-25T15:59:13Z","abstract_excerpt":"Transformer architecture has shown impressive performance in multiple research domains and has become the backbone of many neural network models. However, there is limited understanding on how it works. In particular, with a simple predictive loss, how the representation emerges from the gradient \\emph{training dynamics} remains a mystery. In this paper, for 1-layer transformer with one self-attention layer plus one decoder layer, we analyze its SGD training dynamics for the task of next token prediction in a mathematically rigorous manner. We open the black box of the dynamic process of how t"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2305.16380","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2305.16380/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2305.16380","created_at":"2026-07-05T07:06:27.995662+00:00"},{"alias_kind":"arxiv_version","alias_value":"2305.16380v4","created_at":"2026-07-05T07:06:27.995662+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2305.16380","created_at":"2026-07-05T07:06:27.995662+00:00"},{"alias_kind":"pith_short_12","alias_value":"CTQW7ROD4YLK","created_at":"2026-07-05T07:06:27.995662+00:00"},{"alias_kind":"pith_short_16","alias_value":"CTQW7ROD4YLKO5J7","created_at":"2026-07-05T07:06:27.995662+00:00"},{"alias_kind":"pith_short_8","alias_value":"CTQW7ROD","created_at":"2026-07-05T07:06:27.995662+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.09731","citing_title":"Tight Sample Complexity of Transformers","ref_index":38,"is_internal_anchor":false},{"citing_arxiv_id":"2402.02750","citing_title":"KIVI: A Tuning-Free Asymmetric 2bit Quantization for KV Cache","ref_index":17,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/CTQW7ROD4YLKO5J77TJWVWGMZX","json":"https://pith.science/pith/CTQW7ROD4YLKO5J77TJWVWGMZX.json","graph_json":"https://pith.science/api/pith-number/CTQW7ROD4YLKO5J77TJWVWGMZX/graph.json","events_json":"https://pith.science/api/pith-number/CTQW7ROD4YLKO5J77TJWVWGMZX/events.json","paper":"https://pith.science/paper/CTQW7ROD"},"agent_actions":{"view_html":"https://pith.science/pith/CTQW7ROD4YLKO5J77TJWVWGMZX","download_json":"https://pith.science/pith/CTQW7ROD4YLKO5J77TJWVWGMZX.json","view_paper":"https://pith.science/paper/CTQW7ROD","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2305.16380&json=true","fetch_graph":"https://pith.science/api/pith-number/CTQW7ROD4YLKO5J77TJWVWGMZX/graph.json","fetch_events":"https://pith.science/api/pith-number/CTQW7ROD4YLKO5J77TJWVWGMZX/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/CTQW7ROD4YLKO5J77TJWVWGMZX/action/timestamp_anchor","attest_storage":"https://pith.science/pith/CTQW7ROD4YLKO5J77TJWVWGMZX/action/storage_attestation","attest_author":"https://pith.science/pith/CTQW7ROD4YLKO5J77TJWVWGMZX/action/author_attestation","sign_citation":"https://pith.science/pith/CTQW7ROD4YLKO5J77TJWVWGMZX/action/citation_signature","submit_replication":"https://pith.science/pith/CTQW7ROD4YLKO5J77TJWVWGMZX/action/replication_record"}},"created_at":"2026-07-05T07:06:27.995662+00:00","updated_at":"2026-07-05T07:06:27.995662+00:00"}