{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:YISOJUW6QNGAJ6RJXGU66VNG2F","short_pith_number":"pith:YISOJUW6","schema_version":"1.0","canonical_sha256":"c224e4d2de834c04fa29b9a9ef55a6d142cf81218284d72f3e92d8208162f577","source":{"kind":"arxiv","id":"2409.17625","version":3},"attestation_state":"computed","paper":{"title":"Benign Overfitting in Token Selection of Attention Mechanism","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Issei Sato, Keitaro Sakamoto","submitted_at":"2024-09-26T08:20:05Z","abstract_excerpt":"Attention mechanism is a fundamental component of the transformer model and plays a significant role in its success. However, the theoretical understanding of how attention learns to select tokens is still an emerging area of research. In this work, we study the training dynamics and generalization ability of the attention mechanism under classification problems with label noise. We show that, with the characterization of signal-to-noise ratio (SNR), the token selection of attention mechanism achieves benign overfitting, i.e., maintaining high generalization performance despite fitting label n"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2409.17625","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-09-26T08:20:05Z","cross_cats_sorted":[],"title_canon_sha256":"867b41e0e0b86f0d1ec5989d58676b1ec0d410cfb42617059ac36b3d1dc7777c","abstract_canon_sha256":"5cd9dc8e3f8f8845fc05eb16488e52a9f21a05b01412c0f96dfbc6d8ee4d0f0b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:04:40.051929Z","signature_b64":"MEdpKM67Om/Wf4L9JSTUGjxzX6YNMohzKyqNmjSmpcXcRqclQkSdysK99LLeSjGk2nRnDjfWYlClNRejs4FwAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"c224e4d2de834c04fa29b9a9ef55a6d142cf81218284d72f3e92d8208162f577","last_reissued_at":"2026-07-05T11:04:40.051495Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:04:40.051495Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Benign Overfitting in Token Selection of Attention Mechanism","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Issei Sato, Keitaro Sakamoto","submitted_at":"2024-09-26T08:20:05Z","abstract_excerpt":"Attention mechanism is a fundamental component of the transformer model and plays a significant role in its success. However, the theoretical understanding of how attention learns to select tokens is still an emerging area of research. In this work, we study the training dynamics and generalization ability of the attention mechanism under classification problems with label noise. We show that, with the characterization of signal-to-noise ratio (SNR), the token selection of attention mechanism achieves benign overfitting, i.e., maintaining high generalization performance despite fitting label n"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2409.17625","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2409.17625/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2409.17625","created_at":"2026-07-05T11:04:40.051553+00:00"},{"alias_kind":"arxiv_version","alias_value":"2409.17625v3","created_at":"2026-07-05T11:04:40.051553+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2409.17625","created_at":"2026-07-05T11:04:40.051553+00:00"},{"alias_kind":"pith_short_12","alias_value":"YISOJUW6QNGA","created_at":"2026-07-05T11:04:40.051553+00:00"},{"alias_kind":"pith_short_16","alias_value":"YISOJUW6QNGAJ6RJ","created_at":"2026-07-05T11:04:40.051553+00:00"},{"alias_kind":"pith_short_8","alias_value":"YISOJUW6","created_at":"2026-07-05T11:04:40.051553+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.06314","citing_title":"When Does $\\ell_2$-Boosting Overfit Benignly? High-Dimensional Risk Asymptotics and the $\\ell_1$ Implicit Bias","ref_index":64,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06314","citing_title":"When Does $\\ell_2$-Boosting Overfit Benignly? High-Dimensional Risk Asymptotics and the $\\ell_1$ Implicit Bias","ref_index":64,"is_internal_anchor":false},{"citing_arxiv_id":"2604.19724","citing_title":"Benign Overfitting in Adversarial Training for Vision Transformers","ref_index":66,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/YISOJUW6QNGAJ6RJXGU66VNG2F","json":"https://pith.science/pith/YISOJUW6QNGAJ6RJXGU66VNG2F.json","graph_json":"https://pith.science/api/pith-number/YISOJUW6QNGAJ6RJXGU66VNG2F/graph.json","events_json":"https://pith.science/api/pith-number/YISOJUW6QNGAJ6RJXGU66VNG2F/events.json","paper":"https://pith.science/paper/YISOJUW6"},"agent_actions":{"view_html":"https://pith.science/pith/YISOJUW6QNGAJ6RJXGU66VNG2F","download_json":"https://pith.science/pith/YISOJUW6QNGAJ6RJXGU66VNG2F.json","view_paper":"https://pith.science/paper/YISOJUW6","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2409.17625&json=true","fetch_graph":"https://pith.science/api/pith-number/YISOJUW6QNGAJ6RJXGU66VNG2F/graph.json","fetch_events":"https://pith.science/api/pith-number/YISOJUW6QNGAJ6RJXGU66VNG2F/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/YISOJUW6QNGAJ6RJXGU66VNG2F/action/timestamp_anchor","attest_storage":"https://pith.science/pith/YISOJUW6QNGAJ6RJXGU66VNG2F/action/storage_attestation","attest_author":"https://pith.science/pith/YISOJUW6QNGAJ6RJXGU66VNG2F/action/author_attestation","sign_citation":"https://pith.science/pith/YISOJUW6QNGAJ6RJXGU66VNG2F/action/citation_signature","submit_replication":"https://pith.science/pith/YISOJUW6QNGAJ6RJXGU66VNG2F/action/replication_record"}},"created_at":"2026-07-05T11:04:40.051553+00:00","updated_at":"2026-07-05T11:04:40.051553+00:00"}