{"as_of":"2026-08-08T01:10:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:c3bb7276842eef8c5346dde7e9bc15f895b49da74fcc55c4b96d0e6ca9ca154c","coverage":[{"denominator":56,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":56,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T06:10:21.549139Z","state":"measured"},{"denominator":56,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":56,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-07T06:34:17.273281+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2506.06179/citation-record","integrity":"/paper/2506.06179/integrity","json":"/paper/2506.06179/citation-record.json","paper":"/paper/2506.06179"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2306.00297","last_updated":"2023-11-09T21:46:18Z","snapshot_observed_at":"2026-07-06T15:36:19.023553Z","submitted_at":"2023-06-01T02:35:57Z","title":"Transformers learn to implement preconditioned gradient descent for in-context learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.00297","snapshot_observed_at":"2026-08-07T06:10:16.045295Z","title":"Transformers learn to implement preconditioned gradient descent for in-context learning, November 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.06179","last_updated":"2025-06-06T15:44:10Z","snapshot_observed_at":"2026-08-07T08:23:24.866121Z","submitted_at":"2025-06-06T15:44:10Z","title":"A Theoretical Study of (Hyper) Self-Attention through the Lens of Interactions: Representation, Training, Generalization","version":1},"reference_index":1,"source":"arxiv_source","source_observed_at":"2026-08-07T06:10:16.045295Z"},"links":{"cited_paper":"/paper/2306.00297","citing_paper":"/paper/2506.06179"},"observation_digest":"sha256:ac9117f7ead668c6397848fd9591225b1bad0d996529273ec4cb68eff381b4a4","observation_id":"b186e2f3-97c6-49a0-869d-367bed3f96c2","resolution":{"observed_at":"2026-08-07T06:10:16.045295Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.01082","last_updated":"2024-03-13T16:48:27Z","snapshot_observed_at":"2026-08-02T11:56:02.354026Z","submitted_at":"2023-10-02T10:48:42Z","title":"Linear attention is (maybe) all you need (to understand transformer optimization)","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.01082","snapshot_observed_at":"2026-08-07T06:10:16.118380Z","title":"Linear attention is (maybe) all you need (to understand transformer optimization)","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.06179","last_updated":"2025-06-06T15:44:10Z","snapshot_observed_at":"2026-08-07T08:23:24.866121Z","submitted_at":"2025-06-06T15:44:10Z","title":"A Theoretical Study of (Hyper) Self-Attention through the Lens of Interactions: Representation, Training, Generalization","version":1},"reference_index":2,"source":"arxiv_source","source_observed_at":"2026-08-07T06:10:16.118380Z"},"links":{"cited_paper":"/paper/2310.01082","citing_paper":"/paper/2506.06179"},"observation_digest":"sha256:325311e9db0b90b752f9dbd48b7c7ea568b736c3abd11489676b3389d3d55002","observation_id":"2829d5ae-d210-43c7-a57f-8b0f2fb32faf","resolution":{"observed_at":"2026-08-07T06:10:16.118380Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T06:10:25.496499Z","title":"Block coordinate descent for neural networks provably finds global minima","venue":null,"work_id":"75349709-475c-4050-a37b-7be0e47ab73a","year":2024},"citing_paper":{"arxiv_id":"2506.06179","last_updated":"2025-06-06T15:44:10Z","snapshot_observed_at":"2026-08-07T08:23:24.866121Z","submitted_at":"2025-06-06T15:44:10Z","title":"A Theoretical Study of (Hyper) Self-Attention through the Lens of Interactions: Representation, Training, Generalization","version":1},"reference_index":3,"source":"arxiv_source","source_observed_at":"2026-08-07T06:10:16.308487Z"},"links":{"citing_paper":"/paper/2506.06179"},"observation_digest":"sha256:5b09fb1b85d2c0e58e0b1d4be9c4565f261e6fe13615adef3324b72dfcba6382","observation_id":"b86348a3-4561-4195-b6ea-de5fbd1c5f02","resolution":{"observed_at":"2026-08-07T06:10:25.632599Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.04064","last_updated":"2023-10-06T07:42:39Z","snapshot_observed_at":"2026-08-07T22:03:56.533359Z","submitted_at":"2023-10-06T07:42:39Z","title":"How to Capture Higher-order Correlations? Generalizing Matrix Softmax Attention to Kronecker Computation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.04064","snapshot_observed_at":"2026-08-07T06:10:16.430307Z","title":"and Song, Z","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.06179","last_updated":"2025-06-06T15:44:10Z","snapshot_observed_at":"2026-08-07T08:23:24.866121Z","submitted_at":"2025-06-06T15:44:10Z","title":"A Theoretical Study of (Hyper) Self-Attention through the Lens of Interactions: Representation, Training, Generalization","version":1},"reference_index":4,"source":"arxiv_source","source_observed_at":"2026-08-07T06:10:16.430307Z"},"links":{"cited_paper":"/paper/2310.04064","citing_paper":"/paper/2506.06179"},"observation_digest":"sha256:4ef29b7a10fee79e797cfbb08832be5cd913a8fb288f8dbbc6777f807caac4c4","observation_id":"ed9c1163-000a-4186-bcff-5fc84f90fc4f","resolution":{"observed_at":"2026-08-07T06:10:16.430307Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1409.0473","last_updated":"2016-05-19T21:53:22Z","snapshot_observed_at":"2026-07-06T03:53:10.336430Z","submitted_at":"2014-09-01T16:33:02Z","title":"Neural Machine Translation by Jointly Learning to Align and Translate","version":7},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1409.0473","snapshot_observed_at":"2026-08-07T06:10:16.520290Z","title":"Neural Machine Translation by Jointly Learning to Align and Translate , May 2016","venue":null,"work_id":null,"year":2016},"citing_paper":{"arxiv_id":"2506.06179","last_updated":"2025-06-06T15:44:10Z","snapshot_observed_at":"2026-08-07T08:23:24.866121Z","submitted_at":"2025-06-06T15:44:10Z","title":"A Theoretical Study of (Hyper) Self-Attention through the Lens of Interactions: Representation, Training, Generalization","version":1},"reference_index":5,"source":"arxiv_source","source_observed_at":"2026-08-07T06:10:16.520290Z"},"links":{"cited_paper":"/paper/1409.0473","citing_paper":"/paper/2506.06179"},"observation_digest":"sha256:e7c7fbea8078640d47ed95711abd32e4708b37a256349a892c9186516bb71586","observation_id":"e07cc56c-df49-44ff-8f9e-95ae7a63a9e7","resolution":{"observed_at":"2026-08-07T06:10:16.520290Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2009.11264","last_updated":"2020-10-08T12:55:37Z","snapshot_observed_at":"2026-07-06T09:58:11.932794Z","submitted_at":"2020-09-23T17:21:33Z","title":"On the Ability and Limitations of Transformers to Recognize Formal Languages","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2009.11264","snapshot_observed_at":"2026-08-07T06:10:16.604932Z","title":"On the ability and limitations of transformers to recognize formal languages, 2020 a","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2506.06179","last_updated":"2025-06-06T15:44:10Z","snapshot_observed_at":"2026-08-07T08:23:24.866121Z","submitted_at":"2025-06-06T15:44:10Z","title":"A Theoretical Study of (Hyper) Self-Attention through the Lens of Interactions: Representation, Training, Generalization","version":1},"reference_index":6,"source":"arxiv_source","source_observed_at":"2026-08-07T06:10:16.604932Z"},"links":{"cited_paper":"/paper/2009.11264","citing_paper":"/paper/2506.06179"},"observation_digest":"sha256:8af5150ba8a4f584f91fd0ea3194ec5d5765cd118d32eb43926f0ef99491d369","observation_id":"10cf59d9-00e8-4344-8bf5-a1783691c633","resolution":{"observed_at":"2026-08-07T06:10:16.604932Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2006.09286","last_updated":"2020-10-10T13:34:20Z","snapshot_observed_at":"2026-08-07T08:43:04.975683Z","submitted_at":"2020-06-16T16:27:56Z","title":"On the Computational Power of Transformers and its Implications in Sequence Modeling","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2006.09286","snapshot_observed_at":"2026-08-07T06:10:16.691808Z","title":"On the Computational Power of Transformers and its Implications in Sequence Modeling , October 2020 b","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2506.06179","last_updated":"2025-06-06T15:44:10Z","snapshot_observed_at":"2026-08-07T08:23:24.866121Z","submitted_at":"2025-06-06T15:44:10Z","title":"A Theoretical Study of (Hyper) Self-Attention through the Lens of Interactions: Representation, Training, Generalization","version":1},"reference_index":7,"source":"arxiv_source","source_observed_at":"2026-08-07T06:10:16.691808Z"},"links":{"cited_paper":"/paper/2006.09286","citing_paper":"/paper/2506.06179"},"observation_digest":"sha256:e106549d9535bbf3ebde9bdcbed05f3ec14516c741c7d7435fffe0df48b97e88","observation_id":"3da02f76-adfb-40e5-a345-516b08524857","resolution":{"observed_at":"2026-08-07T06:10:16.691808Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T06:10:25.217859Z","title":null,"venue":null,"work_id":"a6eadd3c-4904-4060-a8a2-4837da30683c","year":1901},"citing_paper":{"arxiv_id":"2506.06179","last_updated":"2025-06-06T15:44:10Z","snapshot_observed_at":"2026-08-07T08:23:24.866121Z","submitted_at":"2025-06-06T15:44:10Z","title":"A Theoretical Study of (Hyper) Self-Attention through the Lens of Interactions: Representation, Training, Generalization","version":1},"reference_index":8,"source":"arxiv_source","source_observed_at":"2026-08-07T06:10:16.756489Z"},"links":{"citing_paper":"/paper/2506.06179"},"observation_digest":"sha256:2dd5c0cf09122da12fbc1357e532cd92e2d52037cc5a37f7fa04c9e096e5f2ed","observation_id":"7e95b74c-6263-4093-bc2a-4b82c380afc1","resolution":{"observed_at":"2026-08-07T06:10:25.347936Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T06:10:16.825748Z","title":null,"venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2506.06179","last_updated":"2025-06-06T15:44:10Z","snapshot_observed_at":"2026-08-07T08:23:24.866121Z","submitted_at":"2025-06-06T15:44:10Z","title":"A Theoretical Study of (Hyper) Self-Attention through the Lens of Interactions: Representation, Training, Generalization","version":1},"reference_index":9,"source":"arxiv_source","source_observed_at":"2026-08-07T06:10:16.825748Z"},"links":{"citing_paper":"/paper/2506.06179"},"observation_digest":"sha256:ce3fe5256035f40f0e1460482c82b8959f58f442edb39095bfcc89b189ab9ae1","observation_id":"d1e61cd4-1248-4fe4-9c20-fc4aaaf28195","resolution":{"observed_at":"2026-08-07T06:10:16.825748Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2106.01345","last_updated":"2021-06-24T17:09:59Z","snapshot_observed_at":"2026-08-07T09:11:20.723647Z","submitted_at":"2021-06-02T17:53:39Z","title":"Decision Transformer: Reinforcement Learning via Sequence Modeling","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2106.01345","snapshot_observed_at":"2026-08-07T06:10:16.886542Z","title":"Decision Transformer : Reinforcement Learning via Sequence Modeling , June 2021","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2506.06179","last_updated":"2025-06-06T15:44:10Z","snapshot_observed_at":"2026-08-07T08:23:24.866121Z","submitted_at":"2025-06-06T15:44:10Z","title":"A Theoretical Study of (Hyper) Self-Attention through the Lens of Interactions: Representation, Training, Generalization","version":1},"reference_index":10,"source":"arxiv_source","source_observed_at":"2026-08-07T06:10:16.886542Z"},"links":{"cited_paper":"/paper/2106.01345","citing_paper":"/paper/2506.06179"},"observation_digest":"sha256:d0918fd47f3d159b8696a38640bd32f319dc1516930aedd5f0cd4924b7005f56","observation_id":"536d2bdd-a47a-4ba5-935b-198c6aa00d06","resolution":{"observed_at":"2026-08-07T06:10:16.886542Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.04084","last_updated":"2024-02-06T15:39:09Z","snapshot_observed_at":"2026-07-06T17:26:19.028623Z","submitted_at":"2024-02-06T15:39:09Z","title":"Provably learning a multi-head attention layer","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.04084","snapshot_observed_at":"2026-08-07T06:10:16.973138Z","title":"and Li, Y","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.06179","last_updated":"2025-06-06T15:44:10Z","snapshot_observed_at":"2026-08-07T08:23:24.866121Z","submitted_at":"2025-06-06T15:44:10Z","title":"A Theoretical Study of (Hyper) Self-Attention through the Lens of Interactions: Representation, Training, Generalization","version":1},"reference_index":11,"source":"arxiv_source","source_observed_at":"2026-08-07T06:10:16.973138Z"},"links":{"cited_paper":"/paper/2402.04084","citing_paper":"/paper/2506.06179"},"observation_digest":"sha256:8396dbfcb610b811680f9a71015db38f94243c86d9aef1283e28a88b5af46f10","observation_id":"56b1e1c8-31ae-4eb2-a06f-9c493c7e724e","resolution":{"observed_at":"2026-08-07T06:10:16.973138Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.19442","last_updated":"2024-06-10T17:18:07Z","snapshot_observed_at":"2026-08-07T21:58:57.595600Z","submitted_at":"2024-02-29T18:43:52Z","title":"Training Dynamics of Multi-Head Softmax Attention for In-Context Learning: Emergence, Convergence, and Optimality","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.19442","snapshot_observed_at":"2026-08-07T06:10:17.068890Z","title":"Training dynamics of multi-head softmax attention for in-context learning: Emergence, convergence, and optimality","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.06179","last_updated":"2025-06-06T15:44:10Z","snapshot_observed_at":"2026-08-07T08:23:24.866121Z","submitted_at":"2025-06-06T15:44:10Z","title":"A Theoretical Study of (Hyper) Self-Attention through the Lens of Interactions: Representation, Training, Generalization","version":1},"reference_index":12,"source":"arxiv_source","source_observed_at":"2026-08-07T06:10:17.068890Z"},"links":{"cited_paper":"/paper/2402.19442","citing_paper":"/paper/2506.06179"},"observation_digest":"sha256:25c0dde1207e9ddb64776638edeec53f394e7a8dfd8325a79cfcdbb7370d791d","observation_id":"3c02b589-bf51-4c11-a0ad-75d7b76b6c0f","resolution":{"observed_at":"2026-08-07T06:10:17.068890Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.06528","last_updated":"2024-06-04T00:20:05Z","snapshot_observed_at":"2026-07-06T16:59:56.350207Z","submitted_at":"2023-12-11T17:05:25Z","title":"Transformers Implement Functional Gradient Descent to Learn Non-Linear Functions In Context","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.06528","snapshot_observed_at":"2026-08-07T06:10:17.138258Z","title":"Transformers implement functional gradient descent to learn non-linear functions in context","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.06179","last_updated":"2025-06-06T15:44:10Z","snapshot_observed_at":"2026-08-07T08:23:24.866121Z","submitted_at":"2025-06-06T15:44:10Z","title":"A Theoretical Study of (Hyper) Self-Attention through the Lens of Interactions: Representation, Training, Generalization","version":1},"reference_index":13,"source":"arxiv_source","source_observed_at":"2026-08-07T06:10:17.138258Z"},"links":{"cited_paper":"/paper/2312.06528","citing_paper":"/paper/2506.06179"},"observation_digest":"sha256:da13bde78a49310f88c106a1e95386b8dd16033fa728d9c6135acb7e1a839e3e","observation_id":"ecfbb607-d2bd-4d8a-bb96-9b0fc2788508","resolution":{"observed_at":"2026-08-07T06:10:17.138258Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2009.14794","last_updated":"2022-11-19T12:45:21Z","snapshot_observed_at":"2026-08-07T11:26:24.987920Z","submitted_at":"2020-09-30T17:09:09Z","title":"Rethinking Attention with Performers","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2009.14794","snapshot_observed_at":"2026-08-07T06:10:17.231632Z","title":"Rethinking Attention with Performers , November 2022","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2506.06179","last_updated":"2025-06-06T15:44:10Z","snapshot_observed_at":"2026-08-07T08:23:24.866121Z","submitted_at":"2025-06-06T15:44:10Z","title":"A Theoretical Study of (Hyper) Self-Attention through the Lens of Interactions: Representation, Training, Generalization","version":1},"reference_index":14,"source":"arxiv_source","source_observed_at":"2026-08-07T06:10:17.231632Z"},"links":{"cited_paper":"/paper/2009.14794","citing_paper":"/paper/2506.06179"},"observation_digest":"sha256:cf30b7767f880bd1eb3cd15bbd2592b0729ed59f6c02bd2cf39cfa6c466f54f9","observation_id":"6a550f40-a8a8-4ee9-b9cd-7b0e0e2fd991","resolution":{"observed_at":"2026-08-07T06:10:17.231632Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.12680","last_updated":"2024-10-12T04:12:31Z","snapshot_observed_at":"2026-07-06T16:35:35.755643Z","submitted_at":"2023-10-19T12:18:24Z","title":"On the Optimization and Generalization of Multi-head Attention","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.12680","snapshot_observed_at":"2026-08-07T06:10:17.345237Z","title":"On the Optimization and Generalization of Multi -head Attention , October 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.06179","last_updated":"2025-06-06T15:44:10Z","snapshot_observed_at":"2026-08-07T08:23:24.866121Z","submitted_at":"2025-06-06T15:44:10Z","title":"A Theoretical Study of (Hyper) Self-Attention through the Lens of Interactions: Representation, Training, Generalization","version":1},"reference_index":15,"source":"arxiv_source","source_observed_at":"2026-08-07T06:10:17.345237Z"},"links":{"cited_paper":"/paper/2310.12680","citing_paper":"/paper/2506.06179"},"observation_digest":"sha256:dd48d0b40da1e10ea6a266b878887dbe0fa2711ef92e4bbef5354fab425ac641","observation_id":"cd45c953-249f-42de-9f57-f256a71e4bd0","resolution":{"observed_at":"2026-08-07T06:10:17.345237Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1810.04805","last_updated":"2019-05-24T20:37:26Z","snapshot_observed_at":"2026-07-30T09:12:38.100527Z","submitted_at":"2018-10-11T00:50:01Z","title":"BERT: Pre-training of Deep Bidirectional Transformers for Language Understanding","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1810.04805","snapshot_observed_at":"2026-08-07T06:10:17.408915Z","title":"BERT : Pre -training of Deep Bidirectional Transformers for Language Understanding , May 2019","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2506.06179","last_updated":"2025-06-06T15:44:10Z","snapshot_observed_at":"2026-08-07T08:23:24.866121Z","submitted_at":"2025-06-06T15:44:10Z","title":"A Theoretical Study of (Hyper) Self-Attention through the Lens of Interactions: Representation, Training, Generalization","version":1},"reference_index":16,"source":"arxiv_source","source_observed_at":"2026-08-07T06:10:17.408915Z"},"links":{"cited_paper":"/paper/1810.04805","citing_paper":"/paper/2506.06179"},"observation_digest":"sha256:9c35d093b484ab88e24632c34987b3269a9de7387d3c666f2ceff4aeed22c944","observation_id":"d9661c3a-1b2a-4d0c-a859-bb38637c1eae","resolution":{"observed_at":"2026-08-07T06:10:17.408915Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2010.11929","last_updated":"2021-06-03T13:08:56Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2020-10-22T17:55:59Z","title":"An Image is Worth 16x16 Words: Transformers for Image Recognition at Scale","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2010.11929","snapshot_observed_at":"2026-08-07T06:10:17.511563Z","title":"An Image is Worth 16x16 Words : Transformers for Image Recognition at Scale , June 2021","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2506.06179","last_updated":"2025-06-06T15:44:10Z","snapshot_observed_at":"2026-08-07T08:23:24.866121Z","submitted_at":"2025-06-06T15:44:10Z","title":"A Theoretical Study of (Hyper) Self-Attention through the Lens of Interactions: Representation, Training, Generalization","version":1},"reference_index":17,"source":"arxiv_source","source_observed_at":"2026-08-07T06:10:17.511563Z"},"links":{"cited_paper":"/paper/2010.11929","citing_paper":"/paper/2506.06179"},"observation_digest":"sha256:26898c94fbcfc2a41c9c08b5b3b49023e40a914d2383b922e5481de060ef6d06","observation_id":"f99701b9-2adf-497f-9573-589492dc5b38","resolution":{"observed_at":"2026-08-07T06:10:17.511563Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2110.10090","last_updated":"2022-06-24T02:32:42Z","snapshot_observed_at":"2026-08-06T01:33:35.413345Z","submitted_at":"2021-10-19T16:36:19Z","title":"Inductive Biases and Variable Creation in Self-Attention Mechanisms","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2110.10090","snapshot_observed_at":"2026-08-07T06:10:17.582759Z","title":"L., Goel, S., Kakade, S., and Zhang, C","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2506.06179","last_updated":"2025-06-06T15:44:10Z","snapshot_observed_at":"2026-08-07T08:23:24.866121Z","submitted_at":"2025-06-06T15:44:10Z","title":"A Theoretical Study of (Hyper) Self-Attention through the Lens of Interactions: Representation, Training, Generalization","version":1},"reference_index":18,"source":"arxiv_source","source_observed_at":"2026-08-07T06:10:17.582759Z"},"links":{"cited_paper":"/paper/2110.10090","citing_paper":"/paper/2506.06179"},"observation_digest":"sha256:70cea04a07d67435cdf84a842e6baba8637a61f384a9fc680e6e2c2a31117aff","observation_id":"4d4f7717-7ac1-4bc6-b735-738760779656","resolution":{"observed_at":"2026-08-07T06:10:17.582759Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T06:10:17.672307Z","title":"A mathematical framework for transformer circuits","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2506.06179","last_updated":"2025-06-06T15:44:10Z","snapshot_observed_at":"2026-08-07T08:23:24.866121Z","submitted_at":"2025-06-06T15:44:10Z","title":"A Theoretical Study of (Hyper) Self-Attention through the Lens of Interactions: Representation, Training, Generalization","version":1},"reference_index":19,"source":"arxiv_source","source_observed_at":"2026-08-07T06:10:17.672307Z"},"links":{"citing_paper":"/paper/2506.06179"},"observation_digest":"sha256:d1d3802e68d93ac8c27498b62a0cdccf6ccddc445c6aed124b4c2c9fc001cb7c","observation_id":"88ce5238-6b0a-4348-94a3-7f4802a095c6","resolution":{"observed_at":"2026-08-07T06:10:17.672307Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.1007/978-1-4471-5310-8","metadata_source":"openalex","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-08T00:03:56.115653Z","title":"Phenotypes and Genotypes: The Search for Influential Genes, volume 18 of Computational Biology","venue":"Computational biology","work_id":"c66a3425-a4b4-4775-ad7b-15f42d1b3aff","year":2016},"citing_paper":{"arxiv_id":"2506.06179","last_updated":"2025-06-06T15:44:10Z","snapshot_observed_at":"2026-08-07T08:23:24.866121Z","submitted_at":"2025-06-06T15:44:10Z","title":"A Theoretical Study of (Hyper) Self-Attention through the Lens of Interactions: Representation, Training, Generalization","version":1},"reference_index":20,"source":"arxiv_source","source_observed_at":"2026-08-07T06:10:17.767563Z"},"links":{"citing_paper":"/paper/2506.06179"},"observation_digest":"sha256:8ad82dc238432b836731435ca00788f625fca8a4425dcf884a07965e282a5199","observation_id":"31ae237e-f46c-40aa-868c-c80d2b355632","resolution":{"observed_at":"2026-08-07T06:10:22.175983Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T06:10:24.984804Z","title":"M., and Fan, J","venue":null,"work_id":"4fc6dafc-37d3-48f7-a800-d7cdab8678de","year":2024},"citing_paper":{"arxiv_id":"2506.06179","last_updated":"2025-06-06T15:44:10Z","snapshot_observed_at":"2026-08-07T08:23:24.866121Z","submitted_at":"2025-06-06T15:44:10Z","title":"A Theoretical Study of (Hyper) Self-Attention through the Lens of Interactions: Representation, Training, Generalization","version":1},"reference_index":21,"source":"arxiv_source","source_observed_at":"2026-08-07T06:10:17.829721Z"},"links":{"citing_paper":"/paper/2506.06179"},"observation_digest":"sha256:e460d3e246f0ee55172a1ae29b623b69eba169aa04d89b0868f8417acc23cdeb","observation_id":"22bbf9eb-1c4c-4bc6-92ea-a01c9e52edb3","resolution":{"observed_at":"2026-08-07T06:10:25.117893Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.04089","last_updated":"2024-06-06T13:59:51Z","snapshot_observed_at":"2026-07-06T18:26:33.307840Z","submitted_at":"2024-06-06T13:59:51Z","title":"On Limitation of Transformer for Learning HMMs","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.04089","snapshot_observed_at":"2026-08-07T06:10:17.900610Z","title":"On Limitation of Transformer for Learning HMMs , June 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.06179","last_updated":"2025-06-06T15:44:10Z","snapshot_observed_at":"2026-08-07T08:23:24.866121Z","submitted_at":"2025-06-06T15:44:10Z","title":"A Theoretical Study of (Hyper) Self-Attention through the Lens of Interactions: Representation, Training, Generalization","version":1},"reference_index":22,"source":"arxiv_source","source_observed_at":"2026-08-07T06:10:17.900610Z"},"links":{"cited_paper":"/paper/2406.04089","citing_paper":"/paper/2506.06179"},"observation_digest":"sha256:2fc2098afbccfbfe95e335f290aebaa9a01214cb890b5ed1e649a1d3c320c9ca","observation_id":"6362e547-8525-4924-bdcc-060998abbc7a","resolution":{"observed_at":"2026-08-07T06:10:17.900610Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T06:10:24.730519Z","title":"In-context convergence of transformers","venue":null,"work_id":"918c9f14-b28a-49d0-beb9-07fd4a8d96b4","year":2024},"citing_paper":{"arxiv_id":"2506.06179","last_updated":"2025-06-06T15:44:10Z","snapshot_observed_at":"2026-08-07T08:23:24.866121Z","submitted_at":"2025-06-06T15:44:10Z","title":"A Theoretical Study of (Hyper) Self-Attention through the Lens of Interactions: Representation, Training, Generalization","version":1},"reference_index":23,"source":"arxiv_source","source_observed_at":"2026-08-07T06:10:18.012937Z"},"links":{"citing_paper":"/paper/2506.06179"},"observation_digest":"sha256:f1acfeff9147bf7a94363034cde008fe98a6d5e47ea518e9016d1ed92eb3fbce","observation_id":"7a8e3379-4fcd-4c62-a39f-9cf93381c4c3","resolution":{"observed_at":"2026-08-07T06:10:24.856363Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T06:10:24.472918Z","title":"How Transformers Learn Diverse Attention Correlations in Masked Vision Pretraining","venue":null,"work_id":"c1636336-5e63-4e9d-81a8-6bbe52b04370","year":2024},"citing_paper":{"arxiv_id":"2506.06179","last_updated":"2025-06-06T15:44:10Z","snapshot_observed_at":"2026-08-07T08:23:24.866121Z","submitted_at":"2025-06-06T15:44:10Z","title":"A Theoretical Study of (Hyper) Self-Attention through the Lens of Interactions: Representation, Training, Generalization","version":1},"reference_index":24,"source":"arxiv_source","source_observed_at":"2026-08-07T06:10:18.140528Z"},"links":{"citing_paper":"/paper/2506.06179"},"observation_digest":"sha256:b1a11ab6695f42146bb10b97c5f592900643481d09fb6f53f6ae4432003cd5ab","observation_id":"0f2c682a-00e9-44c6-927e-990e382e8067","resolution":{"observed_at":"2026-08-07T06:10:24.576290Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2210.09221","last_updated":"2022-10-13T19:53:56Z","snapshot_observed_at":"2026-08-02T16:44:47.491764Z","submitted_at":"2022-10-13T19:53:56Z","title":"Vision Transformers provably learn spatial structure","version":1},"cited_work":{"arxiv_id":"2210.09221","doi":null,"metadata_source":"pith","pith_arxiv_id":"2210.09221","snapshot_observed_at":"2026-08-07T06:10:23.419873Z","title":"Vision Transformers provably learn spatial structure","venue":"cs.CV","work_id":"27d0049b-8138-4a25-89a7-d12478caa341","year":2022},"citing_paper":{"arxiv_id":"2506.06179","last_updated":"2025-06-06T15:44:10Z","snapshot_observed_at":"2026-08-07T08:23:24.866121Z","submitted_at":"2025-06-06T15:44:10Z","title":"A Theoretical Study of (Hyper) Self-Attention through the Lens of Interactions: Representation, Training, Generalization","version":1},"reference_index":25,"source":"arxiv_source","source_observed_at":"2026-08-07T06:10:18.267665Z"},"links":{"cited_paper":"/paper/2210.09221","citing_paper":"/paper/2506.06179"},"observation_digest":"sha256:5519ced096e1923d512fa122596222d2ed7924d8c63ce5afb793dc4f71b8e670","observation_id":"a492e44c-f911-481a-84ba-313a963d527c","resolution":{"observed_at":"2026-08-07T06:10:23.476399Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.1017/9781108873710","metadata_source":"doi_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T06:10:21.894643Z","title":null,"venue":null,"work_id":"26760519-728a-4e51-a628-7cac1aa2cb3c","year":2022},"citing_paper":{"arxiv_id":"2506.06179","last_updated":"2025-06-06T15:44:10Z","snapshot_observed_at":"2026-08-07T08:23:24.866121Z","submitted_at":"2025-06-06T15:44:10Z","title":"A Theoretical Study of (Hyper) Self-Attention through the Lens of Interactions: Representation, Training, Generalization","version":1},"reference_index":26,"source":"arxiv_source","source_observed_at":"2026-08-07T06:10:18.493685Z"},"links":{"citing_paper":"/paper/2506.06179"},"observation_digest":"sha256:348154d2f3bae13b47f06cd28bdfbb768c4bef14ac385405e3f8bea7733fa043","observation_id":"3bc12942-9cb9-4d32-8cc7-2ca58c640eff","resolution":{"observed_at":"2026-08-07T06:10:21.958846Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T06:10:18.579936Z","title":null,"venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2506.06179","last_updated":"2025-06-06T15:44:10Z","snapshot_observed_at":"2026-08-07T08:23:24.866121Z","submitted_at":"2025-06-06T15:44:10Z","title":"A Theoretical Study of (Hyper) Self-Attention through the Lens of Interactions: Representation, Training, Generalization","version":1},"reference_index":27,"source":"arxiv_source","source_observed_at":"2026-08-07T06:10:18.579936Z"},"links":{"citing_paper":"/paper/2506.06179"},"observation_digest":"sha256:ec8b4d0bb8944bac85dd39bd45a2e77e563af2488d28a64823967d6b2543657a","observation_id":"daa10853-5dcd-416e-a272-2b2af1fbdbb7","resolution":{"observed_at":"2026-08-07T06:10:18.579936Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.01655","last_updated":"2024-03-17T23:35:24Z","snapshot_observed_at":"2026-07-06T16:26:50.402172Z","submitted_at":"2023-10-02T21:39:04Z","title":"PolySketchFormer: Fast Transformers via Sketching Polynomial Kernels","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.01655","snapshot_observed_at":"2026-08-07T06:10:18.702105Z","title":"PolySketchFormer : Fast Transformers via Sketching Polynomial Kernels , March 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.06179","last_updated":"2025-06-06T15:44:10Z","snapshot_observed_at":"2026-08-07T08:23:24.866121Z","submitted_at":"2025-06-06T15:44:10Z","title":"A Theoretical Study of (Hyper) Self-Attention through the Lens of Interactions: Representation, Training, Generalization","version":1},"reference_index":28,"source":"arxiv_source","source_observed_at":"2026-08-07T06:10:18.702105Z"},"links":{"cited_paper":"/paper/2310.01655","citing_paper":"/paper/2506.06179"},"observation_digest":"sha256:dec2ad80e63fc67154d3f6da680d385f244b78bbc1c15619980344a962fa2af9","observation_id":"f43d11c9-bdd9-46cc-ab9f-22c1271550ef","resolution":{"observed_at":"2026-08-07T06:10:18.702105Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T06:10:24.235812Z","title":"and Sato, I","venue":null,"work_id":"d3d91369-534b-4bc0-bd65-7f02d67f7f58","year":2024},"citing_paper":{"arxiv_id":"2506.06179","last_updated":"2025-06-06T15:44:10Z","snapshot_observed_at":"2026-08-07T08:23:24.866121Z","submitted_at":"2025-06-06T15:44:10Z","title":"A Theoretical Study of (Hyper) Self-Attention through the Lens of Interactions: Representation, Training, Generalization","version":1},"reference_index":29,"source":"arxiv_source","source_observed_at":"2026-08-07T06:10:18.824938Z"},"links":{"citing_paper":"/paper/2506.06179"},"observation_digest":"sha256:c4295c76c0c69f5c36a55309171c9793089ec66c122cf82b979ce7445654d62b","observation_id":"1f9a66d0-e2b3-4bc2-8315-fd07ad268ec9","resolution":{"observed_at":"2026-08-07T06:10:24.310452Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2006.16236","last_updated":"2020-08-31T11:09:32Z","snapshot_observed_at":"2026-08-06T23:24:28.908251Z","submitted_at":"2020-06-29T17:55:38Z","title":"Transformers are RNNs: Fast Autoregressive Transformers with Linear Attention","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2006.16236","snapshot_observed_at":"2026-08-07T06:10:18.920713Z","title":"Transformers are RNNs : Fast Autoregressive Transformers with Linear Attention","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2506.06179","last_updated":"2025-06-06T15:44:10Z","snapshot_observed_at":"2026-08-07T08:23:24.866121Z","submitted_at":"2025-06-06T15:44:10Z","title":"A Theoretical Study of (Hyper) Self-Attention through the Lens of Interactions: Representation, Training, Generalization","version":1},"reference_index":30,"source":"arxiv_source","source_observed_at":"2026-08-07T06:10:18.920713Z"},"links":{"cited_paper":"/paper/2006.16236","citing_paper":"/paper/2506.06179"},"observation_digest":"sha256:512c4ee2e77069867a9dfe50fe609673c905811a2be7c3379a0a91e0bba6ed54","observation_id":"690e44ed-757d-411e-868e-ae61c42ee29c","resolution":{"observed_at":"2026-08-07T06:10:18.920713Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2206.08898","last_updated":"2024-03-23T06:43:47Z","snapshot_observed_at":"2026-07-06T13:22:05.813672Z","submitted_at":"2022-06-17T17:15:01Z","title":"SimA: Simple Softmax-free Attention for Vision Transformers","version":2},"cited_work":{"arxiv_id":"2206.08898","doi":null,"metadata_source":"pith","pith_arxiv_id":"2206.08898","snapshot_observed_at":"2026-08-07T06:10:23.260814Z","title":"SimA: Simple Softmax-free Attention for Vision Transformers","venue":"cs.CV","work_id":"1e3e5379-20c7-4841-8a0c-b0b87224cb8a","year":2022},"citing_paper":{"arxiv_id":"2506.06179","last_updated":"2025-06-06T15:44:10Z","snapshot_observed_at":"2026-08-07T08:23:24.866121Z","submitted_at":"2025-06-06T15:44:10Z","title":"A Theoretical Study of (Hyper) Self-Attention through the Lens of Interactions: Representation, Training, Generalization","version":1},"reference_index":31,"source":"arxiv_source","source_observed_at":"2026-08-07T06:10:19.033907Z"},"links":{"cited_paper":"/paper/2206.08898","citing_paper":"/paper/2506.06179"},"observation_digest":"sha256:89b5ba1e600e0e342ec951f3bbe8922802ba02cebca8ce3b8b7aea62e4df6b0a","observation_id":"6a9d517c-5ac8-4e1f-a088-3fbba6efc302","resolution":{"observed_at":"2026-08-07T06:10:23.317828Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2302.06015","last_updated":"2023-11-12T04:36:45Z","snapshot_observed_at":"2026-08-05T03:55:36.468406Z","submitted_at":"2023-02-12T22:12:35Z","title":"A Theoretical Understanding of Shallow Vision Transformers: Learning, Generalization, and Sample Complexity","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2302.06015","snapshot_observed_at":"2026-08-07T06:10:19.105823Z","title":"A theoretical understanding of shallow vision transformers: Learning, generalization, and sample complexity, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.06179","last_updated":"2025-06-06T15:44:10Z","snapshot_observed_at":"2026-08-07T08:23:24.866121Z","submitted_at":"2025-06-06T15:44:10Z","title":"A Theoretical Study of (Hyper) Self-Attention through the Lens of Interactions: Representation, Training, Generalization","version":1},"reference_index":32,"source":"arxiv_source","source_observed_at":"2026-08-07T06:10:19.105823Z"},"links":{"cited_paper":"/paper/2302.06015","citing_paper":"/paper/2506.06179"},"observation_digest":"sha256:77835dfdc139a5174a0ac57d80813675f8c7000751cb0dddef48eb63676eeb4f","observation_id":"4e9af55a-da66-4b5e-b0a8-190a9c3b0978","resolution":{"observed_at":"2026-08-07T06:10:19.105823Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.13276","last_updated":"2023-04-26T04:33:41Z","snapshot_observed_at":"2026-08-07T10:06:04.747785Z","submitted_at":"2023-04-26T04:33:41Z","title":"The Closeness of In-Context Learning and Weight Shifting for Softmax Regression","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.13276","snapshot_observed_at":"2026-08-07T06:10:19.205176Z","title":"The closeness of in-context learning and weight shifting for softmax regression","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.06179","last_updated":"2025-06-06T15:44:10Z","snapshot_observed_at":"2026-08-07T08:23:24.866121Z","submitted_at":"2025-06-06T15:44:10Z","title":"A Theoretical Study of (Hyper) Self-Attention through the Lens of Interactions: Representation, Training, Generalization","version":1},"reference_index":33,"source":"arxiv_source","source_observed_at":"2026-08-07T06:10:19.205176Z"},"links":{"cited_paper":"/paper/2304.13276","citing_paper":"/paper/2506.06179"},"observation_digest":"sha256:f8cd212b77865216c31ce4014c2becd207fc2d218c2276abac80878411651eb5","observation_id":"10d5a435-75ec-4bac-ab9e-e3aad03ec028","resolution":{"observed_at":"2026-08-07T06:10:19.205176Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2106.03764","last_updated":"2021-06-08T10:02:26Z","snapshot_observed_at":"2026-08-06T20:31:07.611292Z","submitted_at":"2021-06-07T16:30:28Z","title":"On the Expressive Power of Self-Attention Matrices","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2106.03764","snapshot_observed_at":"2026-08-07T06:10:19.335874Z","title":"On the Expressive Power of Self - Attention Matrices , June 2021","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2506.06179","last_updated":"2025-06-06T15:44:10Z","snapshot_observed_at":"2026-08-07T08:23:24.866121Z","submitted_at":"2025-06-06T15:44:10Z","title":"A Theoretical Study of (Hyper) Self-Attention through the Lens of Interactions: Representation, Training, Generalization","version":1},"reference_index":34,"source":"arxiv_source","source_observed_at":"2026-08-07T06:10:19.335874Z"},"links":{"cited_paper":"/paper/2106.03764","citing_paper":"/paper/2506.06179"},"observation_digest":"sha256:7c5b7d91d619bf40aa2e56517514a0e1b5e318de7b87f9d9891a141001a1ad80","observation_id":"a19f856e-914a-4962-bedf-4c7694a05efa","resolution":{"observed_at":"2026-08-07T06:10:19.335874Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2210.10749","last_updated":"2023-05-02T14:16:15Z","snapshot_observed_at":"2026-08-05T14:55:09.743440Z","submitted_at":"2022-10-19T17:45:48Z","title":"Transformers Learn Shortcuts to Automata","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2210.10749","snapshot_observed_at":"2026-08-07T06:10:19.422720Z","title":"T., Goel, S., Krishnamurthy, A., and Zhang, C","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.06179","last_updated":"2025-06-06T15:44:10Z","snapshot_observed_at":"2026-08-07T08:23:24.866121Z","submitted_at":"2025-06-06T15:44:10Z","title":"A Theoretical Study of (Hyper) Self-Attention through the Lens of Interactions: Representation, Training, Generalization","version":1},"reference_index":35,"source":"arxiv_source","source_observed_at":"2026-08-07T06:10:19.422720Z"},"links":{"cited_paper":"/paper/2210.10749","citing_paper":"/paper/2506.06179"},"observation_digest":"sha256:a9ff5dd0c1084cd8d2754cdaf2d3fa4bca43eb6fd13550c69004767ee5fae800","observation_id":"4de1ab7e-9de2-452b-9150-05f154133b75","resolution":{"observed_at":"2026-08-07T06:10:19.422720Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.17358","last_updated":"2024-05-30T07:54:40Z","snapshot_observed_at":"2026-07-06T18:20:44.070255Z","submitted_at":"2024-05-27T17:02:35Z","title":"Rethinking Transformers in Solving POMDPs","version":3},"cited_work":{"arxiv_id":"2405.17358","doi":null,"metadata_source":"pith","pith_arxiv_id":"2405.17358","snapshot_observed_at":"2026-08-07T06:10:22.962657Z","title":"Rethinking Transformers in Solving POMDPs","venue":"cs.LG","work_id":"e609ff2c-52c7-4244-9522-7ff7e23508b8","year":2024},"citing_paper":{"arxiv_id":"2506.06179","last_updated":"2025-06-06T15:44:10Z","snapshot_observed_at":"2026-08-07T08:23:24.866121Z","submitted_at":"2025-06-06T15:44:10Z","title":"A Theoretical Study of (Hyper) Self-Attention through the Lens of Interactions: Representation, Training, Generalization","version":1},"reference_index":36,"source":"arxiv_source","source_observed_at":"2026-08-07T06:10:19.505098Z"},"links":{"cited_paper":"/paper/2405.17358","citing_paper":"/paper/2506.06179"},"observation_digest":"sha256:557de8b1c4a3457fd60406b5c8bf14471f0006dc541cbce010ff2aba26f3d38f","observation_id":"4a00c977-f4d6-4b39-9d34-8f98f198fe77","resolution":{"observed_at":"2026-08-07T06:10:23.099785Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T06:10:24.009509Z","title":"Your transformer may not be as powerful as you expect","venue":null,"work_id":"df9c6213-5e6f-4eff-9813-f71d361cf811","year":2022},"citing_paper":{"arxiv_id":"2506.06179","last_updated":"2025-06-06T15:44:10Z","snapshot_observed_at":"2026-08-07T08:23:24.866121Z","submitted_at":"2025-06-06T15:44:10Z","title":"A Theoretical Study of (Hyper) Self-Attention through the Lens of Interactions: Representation, Training, Generalization","version":1},"reference_index":37,"source":"arxiv_source","source_observed_at":"2026-08-07T06:10:19.623244Z"},"links":{"citing_paper":"/paper/2506.06179"},"observation_digest":"sha256:9021cc612a3fef9a6eca954e051f130b1fb91495295eb540f9cf11fc7db087b0","observation_id":"609770b0-6828-4366-8570-e6ee8d580761","resolution":{"observed_at":"2026-08-07T06:10:24.114111Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.15478","last_updated":"2024-08-30T05:02:12Z","snapshot_observed_at":"2026-08-05T13:48:34.837029Z","submitted_at":"2024-02-23T18:12:53Z","title":"Transformers are Expressive, But Are They Expressive Enough for Regression?","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.15478","snapshot_observed_at":"2026-08-07T06:10:19.708151Z","title":"Transformers are expressive, but are they expressive enough for regression? arXiv preprint arXiv:2402.15478, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.06179","last_updated":"2025-06-06T15:44:10Z","snapshot_observed_at":"2026-08-07T08:23:24.866121Z","submitted_at":"2025-06-06T15:44:10Z","title":"A Theoretical Study of (Hyper) Self-Attention through the Lens of Interactions: Representation, Training, Generalization","version":1},"reference_index":38,"source":"arxiv_source","source_observed_at":"2026-08-07T06:10:19.708151Z"},"links":{"cited_paper":"/paper/2402.15478","citing_paper":"/paper/2506.06179"},"observation_digest":"sha256:bba59bac2bae8dad3478152743edb6c9702c34777aabe42c3728d6a6778152c2","observation_id":"eccb2b48-c5b2-4ff5-abaa-3473c20fd9ba","resolution":{"observed_at":"2026-08-07T06:10:19.708151Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2409.04431","last_updated":"2025-01-22T01:18:51Z","snapshot_observed_at":"2026-07-06T19:11:40.618767Z","submitted_at":"2024-09-06T17:53:26Z","title":"Theory, Analysis, and Best Practices for Sigmoid Self-Attention","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.04431","snapshot_observed_at":"2026-08-07T06:10:19.778678Z","title":"Theory, Analysis , and Best Practices for Sigmoid Self - Attention , September 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.06179","last_updated":"2025-06-06T15:44:10Z","snapshot_observed_at":"2026-08-07T08:23:24.866121Z","submitted_at":"2025-06-06T15:44:10Z","title":"A Theoretical Study of (Hyper) Self-Attention through the Lens of Interactions: Representation, Training, Generalization","version":1},"reference_index":39,"source":"arxiv_source","source_observed_at":"2026-08-07T06:10:19.778678Z"},"links":{"cited_paper":"/paper/2409.04431","citing_paper":"/paper/2506.06179"},"observation_digest":"sha256:a9914fd38a4a43179ded0716840a42129ca7cd2c16112ee213c0bf53a7bca420","observation_id":"2737a320-d6ae-4494-b67f-3d9d3bf4cc1d","resolution":{"observed_at":"2026-08-07T06:10:19.778678Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2306.02896","last_updated":"2023-11-16T14:48:16Z","snapshot_observed_at":"2026-08-07T11:02:21.213228Z","submitted_at":"2023-06-05T14:05:04Z","title":"Representational Strengths and Limitations of Transformers","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.02896","snapshot_observed_at":"2026-08-07T06:10:19.882192Z","title":"Representational Strengths and Limitations of Transformers , November 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.06179","last_updated":"2025-06-06T15:44:10Z","snapshot_observed_at":"2026-08-07T08:23:24.866121Z","submitted_at":"2025-06-06T15:44:10Z","title":"A Theoretical Study of (Hyper) Self-Attention through the Lens of Interactions: Representation, Training, Generalization","version":1},"reference_index":40,"source":"arxiv_source","source_observed_at":"2026-08-07T06:10:19.882192Z"},"links":{"cited_paper":"/paper/2306.02896","citing_paper":"/paper/2506.06179"},"observation_digest":"sha256:54ec3514f9d1d1bbed838934df5c3e091bc9e30e82d13326ce4d6c5931d08e5e","observation_id":"9e29a8a8-5a31-4110-b43b-ed1dbccdd87e","resolution":{"observed_at":"2026-08-07T06:10:19.882192Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1910.05874","last_updated":"2020-09-07T21:21:37Z","snapshot_observed_at":"2026-07-06T08:29:10.516150Z","submitted_at":"2019-10-14T00:50:55Z","title":"Effects of Depth, Width, and Initialization: A Convergence Analysis of Layer-wise Training for Deep Linear Neural Networks","version":2},"cited_work":{"arxiv_id":"1910.05874","doi":"10.48550/arxiv.1910.05874","metadata_source":"pith","pith_arxiv_id":"1910.05874","snapshot_observed_at":"2026-08-07T06:16:28.064256Z","title":"Effects of Depth, Width, and Initialization: A Convergence Analysis of Layer-wise Training for Deep Linear Neural Networks","venue":"cs.LG","work_id":"4de9d023-cc9b-45d3-b766-92c411983919","year":2019},"citing_paper":{"arxiv_id":"2506.06179","last_updated":"2025-06-06T15:44:10Z","snapshot_observed_at":"2026-08-07T08:23:24.866121Z","submitted_at":"2025-06-06T15:44:10Z","title":"A Theoretical Study of (Hyper) Self-Attention through the Lens of Interactions: Representation, Training, Generalization","version":1},"reference_index":41,"source":"arxiv_source","source_observed_at":"2026-08-07T06:10:20.014640Z"},"links":{"cited_paper":"/paper/1910.05874","citing_paper":"/paper/2506.06179"},"observation_digest":"sha256:bc82b9fa4aea8bc1e60cfab83237a53e2bc48713568857065e2b2edd0ff8b6d0","observation_id":"a1ee5037-49bf-4eaa-844f-c146f8fa6315","resolution":{"observed_at":"2026-08-07T06:10:21.767656Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T06:10:23.802117Z","title":"Unraveling the gradient descent dynamics of transformers","venue":null,"work_id":"04d56c41-89fb-44ed-b240-60ad35088fad","year":2024},"citing_paper":{"arxiv_id":"2506.06179","last_updated":"2025-06-06T15:44:10Z","snapshot_observed_at":"2026-08-07T08:23:24.866121Z","submitted_at":"2025-06-06T15:44:10Z","title":"A Theoretical Study of (Hyper) Self-Attention through the Lens of Interactions: Representation, Training, Generalization","version":1},"reference_index":42,"source":"arxiv_source","source_observed_at":"2026-08-07T06:10:20.096084Z"},"links":{"citing_paper":"/paper/2506.06179"},"observation_digest":"sha256:c67762a837abe3cca3e486f2e0c687d38f65191feea2b7492253d753c079ea73","observation_id":"db8e1eb0-9a49-4aa4-8bee-0609d5641170","resolution":{"observed_at":"2026-08-07T06:10:23.887917Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2104.09864","last_updated":"2023-11-08T13:36:32Z","snapshot_observed_at":"2026-07-06T11:01:58.137141Z","submitted_at":"2021-04-20T09:54:06Z","title":"RoFormer: Enhanced Transformer with Rotary Position Embedding","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2104.09864","snapshot_observed_at":"2026-08-07T06:10:20.168393Z","title":"RoFormer : Enhanced Transformer with Rotary Position Embedding","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.06179","last_updated":"2025-06-06T15:44:10Z","snapshot_observed_at":"2026-08-07T08:23:24.866121Z","submitted_at":"2025-06-06T15:44:10Z","title":"A Theoretical Study of (Hyper) Self-Attention through the Lens of Interactions: Representation, Training, Generalization","version":1},"reference_index":43,"source":"arxiv_source","source_observed_at":"2026-08-07T06:10:20.168393Z"},"links":{"cited_paper":"/paper/2104.09864","citing_paper":"/paper/2506.06179"},"observation_digest":"sha256:7ae3f6edd60c8dffd228893972a323eb1babf28114f108526601709d5b68f419","observation_id":"67c703c1-8a6b-4f82-be50-7ba0e106a85f","resolution":{"observed_at":"2026-08-07T06:10:20.168393Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.16898","last_updated":"2024-02-22T18:38:14Z","snapshot_observed_at":"2026-07-06T16:12:55.132006Z","submitted_at":"2023-08-31T17:57:50Z","title":"Transformers as Support Vector Machines","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.16898","snapshot_observed_at":"2026-08-07T06:10:20.263719Z","title":"A., Li, Y., Thrampoulidis, C., and Oymak, S","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.06179","last_updated":"2025-06-06T15:44:10Z","snapshot_observed_at":"2026-08-07T08:23:24.866121Z","submitted_at":"2025-06-06T15:44:10Z","title":"A Theoretical Study of (Hyper) Self-Attention through the Lens of Interactions: Representation, Training, Generalization","version":1},"reference_index":44,"source":"arxiv_source","source_observed_at":"2026-08-07T06:10:20.263719Z"},"links":{"cited_paper":"/paper/2308.16898","citing_paper":"/paper/2506.06179"},"observation_digest":"sha256:3ffaacdbc1b0fc0403650966c40b4fa18a2bacbace9eb0bb492189534dda9f78","observation_id":"2b50d96b-e00c-4ba8-9ca7-4a4b3b64e704","resolution":{"observed_at":"2026-08-07T06:10:20.263719Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.16380","last_updated":"2023-10-30T17:32:08Z","snapshot_observed_at":"2026-07-06T15:33:34.832607Z","submitted_at":"2023-05-25T15:59:13Z","title":"Scan and Snap: Understanding Training Dynamics and Token Composition in 1-layer Transformer","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.16380","snapshot_observed_at":"2026-08-07T06:10:20.436184Z","title":"Scan and snap: Understanding training dynamics and token composition in 1-layer transformer, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.06179","last_updated":"2025-06-06T15:44:10Z","snapshot_observed_at":"2026-08-07T08:23:24.866121Z","submitted_at":"2025-06-06T15:44:10Z","title":"A Theoretical Study of (Hyper) Self-Attention through the Lens of Interactions: Representation, Training, Generalization","version":1},"reference_index":45,"source":"arxiv_source","source_observed_at":"2026-08-07T06:10:20.436184Z"},"links":{"cited_paper":"/paper/2305.16380","citing_paper":"/paper/2506.06179"},"observation_digest":"sha256:a2906cbb23fb6ec593c344f1d896df0dd17dbc50954e361031fba18ea4be14cc","observation_id":"ec4bcdc5-53d4-47c7-a369-99fd34080d22","resolution":{"observed_at":"2026-08-07T06:10:20.436184Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1501.01571","last_updated":"2015-01-07T17:46:02Z","snapshot_observed_at":"2026-08-07T08:22:44.472326Z","submitted_at":"2015-01-07T17:46:02Z","title":"An Introduction to Matrix Concentration Inequalities","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1501.01571","snapshot_observed_at":"2026-08-07T06:10:20.545869Z","title":null,"venue":null,"work_id":null,"year":2015},"citing_paper":{"arxiv_id":"2506.06179","last_updated":"2025-06-06T15:44:10Z","snapshot_observed_at":"2026-08-07T08:23:24.866121Z","submitted_at":"2025-06-06T15:44:10Z","title":"A Theoretical Study of (Hyper) Self-Attention through the Lens of Interactions: Representation, Training, Generalization","version":1},"reference_index":46,"source":"arxiv_source","source_observed_at":"2026-08-07T06:10:20.545869Z"},"links":{"cited_paper":"/paper/1501.01571","citing_paper":"/paper/2506.06179"},"observation_digest":"sha256:1f768ded9dd944c8a803e837c5b807cb4a8962027644bd793c60e5812cce510c","observation_id":"c2bc851c-fdb9-4894-8e9b-b6dbb3bdb473","resolution":{"observed_at":"2026-08-07T06:10:20.545869Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1706.03762","last_updated":"2023-08-02T00:41:18Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2017-06-12T17:57:34Z","title":"Attention Is All You Need","version":7},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1706.03762","snapshot_observed_at":"2026-08-07T06:10:20.677748Z","title":"N., Kaiser, L., and Polosukhin, I","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.06179","last_updated":"2025-06-06T15:44:10Z","snapshot_observed_at":"2026-08-07T08:23:24.866121Z","submitted_at":"2025-06-06T15:44:10Z","title":"A Theoretical Study of (Hyper) Self-Attention through the Lens of Interactions: Representation, Training, Generalization","version":1},"reference_index":47,"source":"arxiv_source","source_observed_at":"2026-08-07T06:10:20.677748Z"},"links":{"cited_paper":"/paper/1706.03762","citing_paper":"/paper/2506.06179"},"observation_digest":"sha256:87e5595f819350f8787efd70f44cffde1659c8967580eca23af7b3c5ce45516a","observation_id":"7d115a87-ac16-4df7-9565-0ad748ae8882","resolution":{"observed_at":"2026-08-07T06:10:20.677748Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2212.07677","last_updated":"2023-05-31T08:59:47Z","snapshot_observed_at":"2026-07-31T03:35:59.280975Z","submitted_at":"2022-12-15T09:21:21Z","title":"Transformers learn in-context by gradient descent","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2212.07677","snapshot_observed_at":"2026-08-07T06:10:20.795541Z","title":"Transformers learn in-context by gradient descent, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.06179","last_updated":"2025-06-06T15:44:10Z","snapshot_observed_at":"2026-08-07T08:23:24.866121Z","submitted_at":"2025-06-06T15:44:10Z","title":"A Theoretical Study of (Hyper) Self-Attention through the Lens of Interactions: Representation, Training, Generalization","version":1},"reference_index":48,"source":"arxiv_source","source_observed_at":"2026-08-07T06:10:20.795541Z"},"links":{"cited_paper":"/paper/2212.07677","citing_paper":"/paper/2506.06179"},"observation_digest":"sha256:075efa909faeea6df0ff6a38af75aa628bd1757ff4409f5eddd5f90c79a80a56","observation_id":"0f0c45dd-5f99-4090-af3d-5d32b5d20624","resolution":{"observed_at":"2026-08-07T06:10:20.795541Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2006.04768","last_updated":"2020-06-14T08:15:54Z","snapshot_observed_at":"2026-07-06T09:27:03.809621Z","submitted_at":"2020-06-08T17:37:52Z","title":"Linformer: Self-Attention with Linear Complexity","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2006.04768","snapshot_observed_at":"2026-08-07T06:10:20.871924Z","title":"Z., Khabsa, M., Fang, H., and Ma, H","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2506.06179","last_updated":"2025-06-06T15:44:10Z","snapshot_observed_at":"2026-08-07T08:23:24.866121Z","submitted_at":"2025-06-06T15:44:10Z","title":"A Theoretical Study of (Hyper) Self-Attention through the Lens of Interactions: Representation, Training, Generalization","version":1},"reference_index":49,"source":"arxiv_source","source_observed_at":"2026-08-07T06:10:20.871924Z"},"links":{"cited_paper":"/paper/2006.04768","citing_paper":"/paper/2506.06179"},"observation_digest":"sha256:b15bb7e192d0f4c07c8c5346c287f6359603f24c8a48f918ba0f641c1834ed6c","observation_id":"ce0005a2-a080-41ca-ba0c-86ce5636c97f","resolution":{"observed_at":"2026-08-07T06:10:20.871924Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.06893","last_updated":"2024-06-11T02:15:53Z","snapshot_observed_at":"2026-07-06T18:28:38.016649Z","submitted_at":"2024-06-11T02:15:53Z","title":"Transformers Provably Learn Sparse Token Selection While Fully-Connected Nets Cannot","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.06893","snapshot_observed_at":"2026-08-07T06:10:20.984325Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.06179","last_updated":"2025-06-06T15:44:10Z","snapshot_observed_at":"2026-08-07T08:23:24.866121Z","submitted_at":"2025-06-06T15:44:10Z","title":"A Theoretical Study of (Hyper) Self-Attention through the Lens of Interactions: Representation, Training, Generalization","version":1},"reference_index":50,"source":"arxiv_source","source_observed_at":"2026-08-07T06:10:20.984325Z"},"links":{"cited_paper":"/paper/2406.06893","citing_paper":"/paper/2506.06179"},"observation_digest":"sha256:21f4a1a1ea514ad1e7b5bf0df87cf3245a78151084fa8ddd08592ca1516dfdb9","observation_id":"30eb5080-6990-49a8-882f-2efd1e737368","resolution":{"observed_at":"2026-08-07T06:10:20.984325Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2107.13163","last_updated":"2023-03-30T06:31:06Z","snapshot_observed_at":"2026-07-06T11:33:12.096573Z","submitted_at":"2021-07-28T04:28:55Z","title":"Statistically Meaningful Approximation: a Case Study on Approximating Turing Machines with Transformers","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2107.13163","snapshot_observed_at":"2026-08-07T06:10:21.092976Z","title":"Statistically meaningful approximation: a case study on approximating turing machines with transformers, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.06179","last_updated":"2025-06-06T15:44:10Z","snapshot_observed_at":"2026-08-07T08:23:24.866121Z","submitted_at":"2025-06-06T15:44:10Z","title":"A Theoretical Study of (Hyper) Self-Attention through the Lens of Interactions: Representation, Training, Generalization","version":1},"reference_index":51,"source":"arxiv_source","source_observed_at":"2026-08-07T06:10:21.092976Z"},"links":{"cited_paper":"/paper/2107.13163","citing_paper":"/paper/2506.06179"},"observation_digest":"sha256:07883ff9148c62aff40f8fca9ea29f092352af3e65a98c8a675810e189b25d71","observation_id":"4559da0e-1c67-4936-926e-00f186f34955","resolution":{"observed_at":"2026-08-07T06:10:21.092976Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.09605","last_updated":"2024-10-12T17:50:58Z","snapshot_observed_at":"2026-08-03T13:32:29.359689Z","submitted_at":"2024-10-12T17:50:58Z","title":"Training Dynamics of Transformers to Recognize Word Co-occurrence via Gradient Flow Analysis","version":1},"cited_work":{"arxiv_id":"2410.09605","doi":null,"metadata_source":"pith","pith_arxiv_id":"2410.09605","snapshot_observed_at":"2026-08-07T06:10:22.630026Z","title":"Training Dynamics of Transformers to Recognize Word Co-occurrence via Gradient Flow Analysis","venue":"cs.LG","work_id":"90e08731-8b4e-4450-9275-57da9a47dae1","year":2024},"citing_paper":{"arxiv_id":"2506.06179","last_updated":"2025-06-06T15:44:10Z","snapshot_observed_at":"2026-08-07T08:23:24.866121Z","submitted_at":"2025-06-06T15:44:10Z","title":"A Theoretical Study of (Hyper) Self-Attention through the Lens of Interactions: Representation, Training, Generalization","version":1},"reference_index":52,"source":"arxiv_source","source_observed_at":"2026-08-07T06:10:21.173804Z"},"links":{"cited_paper":"/paper/2410.09605","citing_paper":"/paper/2506.06179"},"observation_digest":"sha256:85d8ef319ab7d25c217497490824a34200db4d115fe34257c439b0a7f2c67032","observation_id":"3c471eee-41ae-4717-a298-d15b0b9f958f","resolution":{"observed_at":"2026-08-07T06:10:22.715533Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2105.11115","last_updated":"2023-03-13T01:47:55Z","snapshot_observed_at":"2026-07-06T11:12:04.788175Z","submitted_at":"2021-05-24T06:42:58Z","title":"Self-Attention Networks Can Process Bounded Hierarchical Languages","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2105.11115","snapshot_observed_at":"2026-08-07T06:10:21.255804Z","title":"Self-attention networks can process bounded hierarchical languages, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.06179","last_updated":"2025-06-06T15:44:10Z","snapshot_observed_at":"2026-08-07T08:23:24.866121Z","submitted_at":"2025-06-06T15:44:10Z","title":"A Theoretical Study of (Hyper) Self-Attention through the Lens of Interactions: Representation, Training, Generalization","version":1},"reference_index":53,"source":"arxiv_source","source_observed_at":"2026-08-07T06:10:21.255804Z"},"links":{"cited_paper":"/paper/2105.11115","citing_paper":"/paper/2506.06179"},"observation_digest":"sha256:e96e2bda1416aec03668f84e359e1a7829cabbc62fbdbdcc7b6ae6faa44f2cbb","observation_id":"65e87cbc-8b73-4884-8d06-560981f90fff","resolution":{"observed_at":"2026-08-07T06:10:21.255804Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1912.10077","last_updated":"2020-02-25T03:12:57Z","snapshot_observed_at":"2026-07-06T08:46:09.705605Z","submitted_at":"2019-12-20T19:49:32Z","title":"Are Transformers universal approximators of sequence-to-sequence functions?","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1912.10077","snapshot_observed_at":"2026-08-07T06:10:21.358940Z","title":"S., Reddi, S","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2506.06179","last_updated":"2025-06-06T15:44:10Z","snapshot_observed_at":"2026-08-07T08:23:24.866121Z","submitted_at":"2025-06-06T15:44:10Z","title":"A Theoretical Study of (Hyper) Self-Attention through the Lens of Interactions: Representation, Training, Generalization","version":1},"reference_index":54,"source":"arxiv_source","source_observed_at":"2026-08-07T06:10:21.358940Z"},"links":{"cited_paper":"/paper/1912.10077","citing_paper":"/paper/2506.06179"},"observation_digest":"sha256:ed44eda78bbea3de002aad41cf498044e21e105a73fe10b8084daa57f3e482e7","observation_id":"e11c36a9-ba77-42ee-894a-2d3458285132","resolution":{"observed_at":"2026-08-07T06:10:21.358940Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1803.00225","last_updated":"2019-05-12T12:24:53Z","snapshot_observed_at":"2026-07-06T06:26:01.372123Z","submitted_at":"2018-03-01T06:11:53Z","title":"Global Convergence of Block Coordinate Descent in Deep Learning","version":4},"cited_work":{"arxiv_id":"1803.00225","doi":null,"metadata_source":"pith","pith_arxiv_id":"1803.00225","snapshot_observed_at":"2026-08-07T06:10:22.440807Z","title":"Global Convergence of Block Coordinate Descent in Deep Learning","venue":"math.OC","work_id":"d20c8486-399e-4adb-ba2e-360ef9251bda","year":2018},"citing_paper":{"arxiv_id":"2506.06179","last_updated":"2025-06-06T15:44:10Z","snapshot_observed_at":"2026-08-07T08:23:24.866121Z","submitted_at":"2025-06-06T15:44:10Z","title":"A Theoretical Study of (Hyper) Self-Attention through the Lens of Interactions: Representation, Training, Generalization","version":1},"reference_index":55,"source":"arxiv_source","source_observed_at":"2026-08-07T06:10:21.459603Z"},"links":{"cited_paper":"/paper/1803.00225","citing_paper":"/paper/2506.06179"},"observation_digest":"sha256:60f303039ab0af933fd2d1c7e22655cfb5be6f0682b112136a28e9045fc32f43","observation_id":"811b1712-1d5b-43dd-8bef-2b8dfe8355f3","resolution":{"observed_at":"2026-08-07T06:10:22.512246Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.14706","last_updated":"2025-01-25T04:05:56Z","snapshot_observed_at":"2026-08-06T21:21:01.700015Z","submitted_at":"2024-10-07T20:31:13Z","title":"Transformers are Efficient Compilers, Provably","version":2},"cited_work":{"arxiv_id":"2410.14706","doi":null,"metadata_source":"pith","pith_arxiv_id":"2410.14706","snapshot_observed_at":"2026-08-07T06:10:22.288094Z","title":"Transformers are Efficient Compilers, Provably","venue":"cs.PL","work_id":"f1672e85-a282-489f-b21a-251cc56b7b06","year":2024},"citing_paper":{"arxiv_id":"2506.06179","last_updated":"2025-06-06T15:44:10Z","snapshot_observed_at":"2026-08-07T08:23:24.866121Z","submitted_at":"2025-06-06T15:44:10Z","title":"A Theoretical Study of (Hyper) Self-Attention through the Lens of Interactions: Representation, Training, Generalization","version":1},"reference_index":56,"source":"arxiv_source","source_observed_at":"2026-08-07T06:10:21.549139Z"},"links":{"cited_paper":"/paper/2410.14706","citing_paper":"/paper/2506.06179"},"observation_digest":"sha256:f57f5c75a4aa8731823f7188bfc81e9c4d71d411f91d3b259a850dc4635f3b3e","observation_id":"1374c329-c8d2-4784-9261-e9b11af7be25","resolution":{"observed_at":"2026-08-07T06:10:22.360182Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2506.06179","last_updated":"2025-06-06T15:44:10Z","latest_version":1,"primary_category":"cs.LG","snapshot_observed_at":"2026-08-07T08:23:24.866121Z","submitted_at":"2025-06-06T15:44:10Z","title":"A Theoretical Study of (Hyper) Self-Attention through the Lens of Interactions: Representation, Training, Generalization"},"reference_resolution":{"displayed":56,"state_counts":{"malformed_identifier":0,"metadata_mismatch":1,"parse_uncertain":0,"unresolved":40,"verified_exact":8,"verified_fuzzy":7},"total_outbound_references":56},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"thesis":"As of 8 August 2026, this Paper Citation Record lists 56 of 56 outbound references and 0 inbound Pith citation observations for arXiv:2506.06179."}