{"as_of":"2026-08-13T13:29:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:e86a63711b90c0898a1fbf54374e91be003a2bbc4d555a21588e06983a253c6b","coverage":[{"denominator":48,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":48,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-12T12:32:45.226075Z","state":"measured"},{"denominator":48,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":48,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-13T06:32:02.005865+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2411.17182/citation-record","integrity":"/paper/2411.17182/integrity","json":"/paper/2411.17182/citation-record.json","paper":"/paper/2411.17182"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T12:32:46.018507Z","title":"Repulsive attention: Rethinking multi-head attention as bayesian inference","venue":null,"work_id":"faac20f7-5a11-49c5-9204-470e67f0c06a","year":2020},"citing_paper":{"arxiv_id":"2411.17182","last_updated":"2024-11-26T07:44:57Z","snapshot_observed_at":"2026-08-12T12:22:29.940094Z","submitted_at":"2024-11-26T07:44:57Z","title":"An In-depth Investigation of Sparse Rate Reduction in Transformer-like Models","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-12T12:32:45.002673Z"},"links":{"citing_paper":"/paper/2411.17182"},"observation_digest":"sha256:ec34d63c3f78db9f4bee3fe71b9724b72e08de2df0f8caf27302e81df75c0722","observation_id":"8ac51476-fb4a-47f2-b0f9-cd2b04cc9ae8","resolution":{"observed_at":"2026-08-12T12:32:46.022978Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1607.06450","last_updated":"2016-07-21T19:57:52Z","snapshot_observed_at":"2026-08-12T08:59:05.030983Z","submitted_at":"2016-07-21T19:57:52Z","title":"Layer Normalization","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1607.06450","snapshot_observed_at":"2026-08-12T12:32:45.007722Z","title":"Layer normalization","venue":null,"work_id":null,"year":2016},"citing_paper":{"arxiv_id":"2411.17182","last_updated":"2024-11-26T07:44:57Z","snapshot_observed_at":"2026-08-12T12:22:29.940094Z","submitted_at":"2024-11-26T07:44:57Z","title":"An In-depth Investigation of Sparse Rate Reduction in Transformer-like Models","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-12T12:32:45.007722Z"},"links":{"cited_paper":"/paper/1607.06450","citing_paper":"/paper/2411.17182"},"observation_digest":"sha256:153c120cb90d1c065e7c650f245677ab4f80c890b84036514396de9ce44d0771","observation_id":"b35fff6e-3134-45ad-a3a2-ce6b93eed7b0","resolution":{"observed_at":"2026-08-12T12:32:45.007722Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T12:32:45.012782Z","title":"Spectrally-normalized margin bounds for neural networks","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2411.17182","last_updated":"2024-11-26T07:44:57Z","snapshot_observed_at":"2026-08-12T12:22:29.940094Z","submitted_at":"2024-11-26T07:44:57Z","title":"An In-depth Investigation of Sparse Rate Reduction in Transformer-like Models","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-12T12:32:45.012782Z"},"links":{"citing_paper":"/paper/2411.17182"},"observation_digest":"sha256:a2087d3e51254e3cab2001c2ef087010edda9eb1b1d5e741b69d346acfdbbd78","observation_id":"6ea967a3-0fb2-4185-baf8-45ae8ea67187","resolution":{"observed_at":"2026-08-12T12:32:45.012782Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T12:32:45.992765Z","title":"Birth of a transformer: A memory viewpoint","venue":null,"work_id":"2a3d03fd-f566-4c05-afcb-11e3bcd197d5","year":2023},"citing_paper":{"arxiv_id":"2411.17182","last_updated":"2024-11-26T07:44:57Z","snapshot_observed_at":"2026-08-12T12:22:29.940094Z","submitted_at":"2024-11-26T07:44:57Z","title":"An In-depth Investigation of Sparse Rate Reduction in Transformer-like Models","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-12T12:32:45.017923Z"},"links":{"citing_paper":"/paper/2411.17182"},"observation_digest":"sha256:0bf1576b1ae01612db8ae5e0bded515e98aa5eb4682ca3133a0ca9cf77d8cc04","observation_id":"5462d016-0753-4c0c-ae96-de693445e988","resolution":{"observed_at":"2026-08-12T12:32:45.998168Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T12:32:45.977261Z","title":"Attention approximates sparse distributed memory","venue":null,"work_id":"61d59f58-9e2a-4cf6-95c6-adbc5d3176be","year":2021},"citing_paper":{"arxiv_id":"2411.17182","last_updated":"2024-11-26T07:44:57Z","snapshot_observed_at":"2026-08-12T12:22:29.940094Z","submitted_at":"2024-11-26T07:44:57Z","title":"An In-depth Investigation of Sparse Rate Reduction in Transformer-like Models","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-12T12:32:45.023011Z"},"links":{"citing_paper":"/paper/2411.17182"},"observation_digest":"sha256:c01854ae37781c65ebaac18a863c0a298c845c4fc12030aeecb602a9992b8a72","observation_id":"9578f20e-5e1e-4c5b-95dc-6e8750604d8b","resolution":{"observed_at":"2026-08-12T12:32:45.982390Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T12:32:45.961051Z","title":"Invariant scattering convolution networks","venue":null,"work_id":"54937322-c14b-47fb-bcee-f62a1ab0ae3f","year":2013},"citing_paper":{"arxiv_id":"2411.17182","last_updated":"2024-11-26T07:44:57Z","snapshot_observed_at":"2026-08-12T12:22:29.940094Z","submitted_at":"2024-11-26T07:44:57Z","title":"An In-depth Investigation of Sparse Rate Reduction in Transformer-like Models","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-12T12:32:45.027526Z"},"links":{"citing_paper":"/paper/2411.17182"},"observation_digest":"sha256:b90ce46922ef4af8117b62a6e68e207194c6da07af3c7166fa4e30d002cd7e2d","observation_id":"0090f6f0-0841-4359-b3ac-4e7327cbb323","resolution":{"observed_at":"2026-08-12T12:32:45.966651Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T12:32:45.033422Z","title":"Emerging properties in self-supervised vision transformers","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2411.17182","last_updated":"2024-11-26T07:44:57Z","snapshot_observed_at":"2026-08-12T12:22:29.940094Z","submitted_at":"2024-11-26T07:44:57Z","title":"An In-depth Investigation of Sparse Rate Reduction in Transformer-like Models","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-12T12:32:45.033422Z"},"links":{"citing_paper":"/paper/2411.17182"},"observation_digest":"sha256:71a972468accbbd046b40f8a84290c3d20ae6f28931df9e8720b6ae0fab34491","observation_id":"3e9edc9b-69b6-42fd-ada6-314db41fb187","resolution":{"observed_at":"2026-08-12T12:32:45.033422Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T12:32:45.935017Z","title":"Transformer interpretability beyond attention visualization","venue":null,"work_id":"439617fe-58e5-4e17-a96b-c3969676d476","year":2021},"citing_paper":{"arxiv_id":"2411.17182","last_updated":"2024-11-26T07:44:57Z","snapshot_observed_at":"2026-08-12T12:22:29.940094Z","submitted_at":"2024-11-26T07:44:57Z","title":"An In-depth Investigation of Sparse Rate Reduction in Transformer-like Models","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-12T12:32:45.038131Z"},"links":{"citing_paper":"/paper/2411.17182"},"observation_digest":"sha256:30ed59898e82d39a62201b601f5c110f7ac773e9362f5fbf5b14d3eabcaed68b","observation_id":"5b9218ed-4289-47eb-bacd-3b04d66e3770","resolution":{"observed_at":"2026-08-12T12:32:45.940227Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T12:32:45.042753Z","title":"Randaugment: Practical automated data augmentation with a reduced search space","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2411.17182","last_updated":"2024-11-26T07:44:57Z","snapshot_observed_at":"2026-08-12T12:22:29.940094Z","submitted_at":"2024-11-26T07:44:57Z","title":"An In-depth Investigation of Sparse Rate Reduction in Transformer-like Models","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-12T12:32:45.042753Z"},"links":{"citing_paper":"/paper/2411.17182"},"observation_digest":"sha256:4c93a596cf963a9509b561dff396a1105a054b702bc9a066d8207908f1b11a14","observation_id":"c11ae912-1203-478c-847b-9370b9337f47","resolution":{"observed_at":"2026-08-12T12:32:45.042753Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T12:32:45.910404Z","title":"Analyzing transformers in embedding space","venue":null,"work_id":"b66e8a0e-38fb-4586-86d1-2ba9f1c48cc5","year":2023},"citing_paper":{"arxiv_id":"2411.17182","last_updated":"2024-11-26T07:44:57Z","snapshot_observed_at":"2026-08-12T12:22:29.940094Z","submitted_at":"2024-11-26T07:44:57Z","title":"An In-depth Investigation of Sparse Rate Reduction in Transformer-like Models","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-12T12:32:45.047548Z"},"links":{"citing_paper":"/paper/2411.17182"},"observation_digest":"sha256:ce9ef701375b6f64161aaefcf32f402812642349afa5b4f4c2700c8d86ebb05b","observation_id":"6b8ff53b-013c-45ee-b6b2-276f5cba3f2e","resolution":{"observed_at":"2026-08-12T12:32:45.915153Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T12:32:45.052104Z","title":"An image is worth 16x16 words: Transformers for image recognition at scale","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2411.17182","last_updated":"2024-11-26T07:44:57Z","snapshot_observed_at":"2026-08-12T12:22:29.940094Z","submitted_at":"2024-11-26T07:44:57Z","title":"An In-depth Investigation of Sparse Rate Reduction in Transformer-like Models","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-12T12:32:45.052104Z"},"links":{"citing_paper":"/paper/2411.17182"},"observation_digest":"sha256:7dc9b367b223b0158939836a54dcfb60f4a0edeb237e33b1f60c0bb2f58d443b","observation_id":"f52c75f1-0776-4cc8-b6a4-d0f8c87c02cd","resolution":{"observed_at":"2026-08-12T12:32:45.052104Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T12:32:45.056671Z","title":"A mathematical framework for transformer circuits","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2411.17182","last_updated":"2024-11-26T07:44:57Z","snapshot_observed_at":"2026-08-12T12:22:29.940094Z","submitted_at":"2024-11-26T07:44:57Z","title":"An In-depth Investigation of Sparse Rate Reduction in Transformer-like Models","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-12T12:32:45.056671Z"},"links":{"citing_paper":"/paper/2411.17182"},"observation_digest":"sha256:5237fd916b021b97eabf388447abb035644f9e52e7af3b417a23f31a9be9ee0f","observation_id":"7d1094ff-ed3a-49a6-8b29-fca4c410bafb","resolution":{"observed_at":"2026-08-12T12:32:45.056671Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T12:32:45.862791Z","title":"The emergence of clusters in self-attention dynamics","venue":null,"work_id":"cb44410f-0f84-455c-90a5-b6529c2067bb","year":2023},"citing_paper":{"arxiv_id":"2411.17182","last_updated":"2024-11-26T07:44:57Z","snapshot_observed_at":"2026-08-12T12:22:29.940094Z","submitted_at":"2024-11-26T07:44:57Z","title":"An In-depth Investigation of Sparse Rate Reduction in Transformer-like Models","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-12T12:32:45.066937Z"},"links":{"citing_paper":"/paper/2411.17182"},"observation_digest":"sha256:7f6dfddef38171de831378d86bbeadc0a3400f63d42f8c369f7e90356ff9b242","observation_id":"c06a0860-7503-4381-acf7-bd362f606f16","resolution":{"observed_at":"2026-08-12T12:32:45.867629Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T12:32:45.847291Z","title":"Patchscopes: A unifying framework for inspecting hidden representations of language models","venue":null,"work_id":"b92c1e15-2e29-4b6b-bebf-524137e496b1","year":2024},"citing_paper":{"arxiv_id":"2411.17182","last_updated":"2024-11-26T07:44:57Z","snapshot_observed_at":"2026-08-12T12:22:29.940094Z","submitted_at":"2024-11-26T07:44:57Z","title":"An In-depth Investigation of Sparse Rate Reduction in Transformer-like Models","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-12T12:32:45.071587Z"},"links":{"citing_paper":"/paper/2411.17182"},"observation_digest":"sha256:798d8dc2ecbf17893086cafcbe0f33d318bbaa95a5fdb8752cee2b61399d1ea1","observation_id":"e79ac14f-aa14-437b-bf24-6cd0f9b3ac1f","resolution":{"observed_at":"2026-08-12T12:32:45.852250Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T12:32:45.831552Z","title":"Learning fast approximations of sparse coding","venue":null,"work_id":"5205cf53-5885-4799-b0ab-e12ed0c955f7","year":2010},"citing_paper":{"arxiv_id":"2411.17182","last_updated":"2024-11-26T07:44:57Z","snapshot_observed_at":"2026-08-12T12:22:29.940094Z","submitted_at":"2024-11-26T07:44:57Z","title":"An In-depth Investigation of Sparse Rate Reduction in Transformer-like Models","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-12T12:32:45.076259Z"},"links":{"citing_paper":"/paper/2411.17182"},"observation_digest":"sha256:c2b9668d0873938b8a6acb98fbf50581a8a1e9a3600d5a364a6bffb83a228f7e","observation_id":"bf391b7c-adc1-4acb-b43b-61c758f08718","resolution":{"observed_at":"2026-08-12T12:32:45.836412Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2212.13345","last_updated":"2022-12-27T02:54:46Z","snapshot_observed_at":"2026-08-13T13:14:04.763006Z","submitted_at":"2022-12-27T02:54:46Z","title":"The Forward-Forward Algorithm: Some Preliminary Investigations","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2212.13345","snapshot_observed_at":"2026-08-12T12:32:45.081163Z","title":"The forward-forward algorithm: Some preliminary investigations","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2411.17182","last_updated":"2024-11-26T07:44:57Z","snapshot_observed_at":"2026-08-12T12:22:29.940094Z","submitted_at":"2024-11-26T07:44:57Z","title":"An In-depth Investigation of Sparse Rate Reduction in Transformer-like Models","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-12T12:32:45.081163Z"},"links":{"cited_paper":"/paper/2212.13345","citing_paper":"/paper/2411.17182"},"observation_digest":"sha256:2e7dc4de41f7dc821e51b3ceea1aeddd90bdd6196dc71c4358ed588c1432bb84","observation_id":"605f79d1-605a-4da1-a9f0-03c135299011","resolution":{"observed_at":"2026-08-12T12:32:45.081163Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T12:32:45.813737Z","title":"Energy transformer","venue":null,"work_id":"bf9b1b98-4615-46a3-9468-4e2058be3985","year":2023},"citing_paper":{"arxiv_id":"2411.17182","last_updated":"2024-11-26T07:44:57Z","snapshot_observed_at":"2026-08-12T12:22:29.940094Z","submitted_at":"2024-11-26T07:44:57Z","title":"An In-depth Investigation of Sparse Rate Reduction in Transformer-like Models","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-12T12:32:45.086133Z"},"links":{"citing_paper":"/paper/2411.17182"},"observation_digest":"sha256:f6eeedc1fa99cf8925219a65400be3c1bd803166b8857dc91c54ac4b9b896a65","observation_id":"1be9dafb-f19b-4cee-a6cc-ca8e4612b738","resolution":{"observed_at":"2026-08-12T12:32:45.819300Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T12:32:45.798006Z","title":"Fantastic generalization measures and where to find them","venue":null,"work_id":"8aea497a-883b-48cc-9866-c3eee41a50dc","year":2020},"citing_paper":{"arxiv_id":"2411.17182","last_updated":"2024-11-26T07:44:57Z","snapshot_observed_at":"2026-08-12T12:22:29.940094Z","submitted_at":"2024-11-26T07:44:57Z","title":"An In-depth Investigation of Sparse Rate Reduction in Transformer-like Models","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-12T12:32:45.090907Z"},"links":{"citing_paper":"/paper/2411.17182"},"observation_digest":"sha256:230fb550a6bbbaa4c8c305da95824743323d994be6ce9168570f094cc20d698a","observation_id":"1625818f-5370-4e2e-a8bb-73ec627ab404","resolution":{"observed_at":"2026-08-12T12:32:45.803266Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T12:32:45.095640Z","title":"A new measure of rank correlation","venue":null,"work_id":null,"year":1938},"citing_paper":{"arxiv_id":"2411.17182","last_updated":"2024-11-26T07:44:57Z","snapshot_observed_at":"2026-08-12T12:22:29.940094Z","submitted_at":"2024-11-26T07:44:57Z","title":"An In-depth Investigation of Sparse Rate Reduction in Transformer-like Models","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-12T12:32:45.095640Z"},"links":{"citing_paper":"/paper/2411.17182"},"observation_digest":"sha256:a37a89c16379d945dd758c2bc751f8a47a5c2f8327c38c524457374a5bec1617","observation_id":"7608b21f-bdd6-41c6-98c5-ab7723c5504e","resolution":{"observed_at":"2026-08-12T12:32:45.095640Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T12:32:45.772154Z","title":"On large-batch training for deep learning: Generalization gap and sharp minima","venue":null,"work_id":"80b3a92c-e859-4e81-95ac-b72a1f08c2b2","year":2016},"citing_paper":{"arxiv_id":"2411.17182","last_updated":"2024-11-26T07:44:57Z","snapshot_observed_at":"2026-08-12T12:22:29.940094Z","submitted_at":"2024-11-26T07:44:57Z","title":"An In-depth Investigation of Sparse Rate Reduction in Transformer-like Models","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-12T12:32:45.100467Z"},"links":{"citing_paper":"/paper/2411.17182"},"observation_digest":"sha256:de2c67371d2230eed898093a6150961679c74f1d9955c417217c248c9f8b2149","observation_id":"8f2f13eb-9e2d-4757-8686-ed2a53edbc07","resolution":{"observed_at":"2026-08-12T12:32:45.776973Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T12:32:45.105058Z","title":"Kingma and Jimmy Ba","venue":null,"work_id":null,"year":2015},"citing_paper":{"arxiv_id":"2411.17182","last_updated":"2024-11-26T07:44:57Z","snapshot_observed_at":"2026-08-12T12:22:29.940094Z","submitted_at":"2024-11-26T07:44:57Z","title":"An In-depth Investigation of Sparse Rate Reduction in Transformer-like Models","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-12T12:32:45.105058Z"},"links":{"citing_paper":"/paper/2411.17182"},"observation_digest":"sha256:4a97bbf5c1abc032e2d08e6fda1b9586597a3c4e4e0505d0625a2ea945c85a08","observation_id":"250b883e-35af-43ae-9acd-9f8f564a1372","resolution":{"observed_at":"2026-08-12T12:32:45.105058Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T12:32:45.746793Z","title":"Tracr: Compiled transformers as a laboratory for interpretability","venue":null,"work_id":"0bb6d549-22e6-44ef-9212-83c45195c4d3","year":2023},"citing_paper":{"arxiv_id":"2411.17182","last_updated":"2024-11-26T07:44:57Z","snapshot_observed_at":"2026-08-12T12:22:29.940094Z","submitted_at":"2024-11-26T07:44:57Z","title":"An In-depth Investigation of Sparse Rate Reduction in Transformer-like Models","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-12T12:32:45.110071Z"},"links":{"citing_paper":"/paper/2411.17182"},"observation_digest":"sha256:cf9fe448bf17296c92766aead09f7be63b1c455dcc10b46bf177f80dc22174ed","observation_id":"76d36dfa-a66a-4e74-ad0c-1ce875852f48","resolution":{"observed_at":"2026-08-12T12:32:45.752778Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T12:32:45.114718Z","title":"Omnigrok: Grokking beyond algorithmic data","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2411.17182","last_updated":"2024-11-26T07:44:57Z","snapshot_observed_at":"2026-08-12T12:22:29.940094Z","submitted_at":"2024-11-26T07:44:57Z","title":"An In-depth Investigation of Sparse Rate Reduction in Transformer-like Models","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-12T12:32:45.114718Z"},"links":{"citing_paper":"/paper/2411.17182"},"observation_digest":"sha256:26cbef16b1b2b8b339754477605810c4d02406eb37b1f41687c04a2901303661","observation_id":"bc6f5695-7460-492c-8500-81222d17031c","resolution":{"observed_at":"2026-08-12T12:32:45.114718Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T12:32:45.720499Z","title":"Segmentation of multivariate mixed data via lossy data coding and compression","venue":null,"work_id":"058e2c88-a06c-4991-a311-9643954693d4","year":2007},"citing_paper":{"arxiv_id":"2411.17182","last_updated":"2024-11-26T07:44:57Z","snapshot_observed_at":"2026-08-12T12:22:29.940094Z","submitted_at":"2024-11-26T07:44:57Z","title":"An In-depth Investigation of Sparse Rate Reduction in Transformer-like Models","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-12T12:32:45.119269Z"},"links":{"citing_paper":"/paper/2411.17182"},"observation_digest":"sha256:47cdba3c11d02334989cf56883c4fb19b521a35a91f6955b0f3bdcb40e343f02","observation_id":"69c8da1a-a61c-4f39-855b-c2171387f5d5","resolution":{"observed_at":"2026-08-12T12:32:45.726347Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T12:32:45.705146Z","title":"Pac-bayesian model averaging","venue":null,"work_id":"2f7895e2-83c6-4b62-9698-5dad099d8018","year":1999},"citing_paper":{"arxiv_id":"2411.17182","last_updated":"2024-11-26T07:44:57Z","snapshot_observed_at":"2026-08-12T12:22:29.940094Z","submitted_at":"2024-11-26T07:44:57Z","title":"An In-depth Investigation of Sparse Rate Reduction in Transformer-like Models","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-12T12:32:45.124143Z"},"links":{"citing_paper":"/paper/2411.17182"},"observation_digest":"sha256:c9d38bbd1c9d9498f3b4ff6de59300e764d5402a4d9118c06141daa9ac9f2994","observation_id":"501377e5-6383-49eb-a7cb-e64f16f231c7","resolution":{"observed_at":"2026-08-12T12:32:45.710222Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T12:32:45.689180Z","title":"Universal hopfield networks: A general framework for single-shot associative memory models","venue":null,"work_id":"f2e83344-2602-4e4a-8b93-3892adc31669","year":2022},"citing_paper":{"arxiv_id":"2411.17182","last_updated":"2024-11-26T07:44:57Z","snapshot_observed_at":"2026-08-12T12:22:29.940094Z","submitted_at":"2024-11-26T07:44:57Z","title":"An In-depth Investigation of Sparse Rate Reduction in Transformer-like Models","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-12T12:32:45.128881Z"},"links":{"citing_paper":"/paper/2411.17182"},"observation_digest":"sha256:42efa47674e072175c25cceb590313beaabf24cea10a7ca046cf7723643c27eb","observation_id":"73fe6ed6-92e0-4665-9c76-4021e94b5bc2","resolution":{"observed_at":"2026-08-12T12:32:45.694650Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T12:32:45.133706Z","title":"Algorithm unrolling: Interpretable, efficient deep learning for signal and image processing","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2411.17182","last_updated":"2024-11-26T07:44:57Z","snapshot_observed_at":"2026-08-12T12:22:29.940094Z","submitted_at":"2024-11-26T07:44:57Z","title":"An In-depth Investigation of Sparse Rate Reduction in Transformer-like Models","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-12T12:32:45.133706Z"},"links":{"citing_paper":"/paper/2411.17182"},"observation_digest":"sha256:8b30e1794af84d88317cc0b9d06baecb060ab544921765e08bfe919dabe7d5df","observation_id":"b50e8e18-e4a0-4c14-8af1-a2d7dc9b730e","resolution":{"observed_at":"2026-08-12T12:32:45.133706Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T12:32:45.138412Z","title":"Progress measures for grokking via mechanistic interpretability","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2411.17182","last_updated":"2024-11-26T07:44:57Z","snapshot_observed_at":"2026-08-12T12:22:29.940094Z","submitted_at":"2024-11-26T07:44:57Z","title":"An In-depth Investigation of Sparse Rate Reduction in Transformer-like Models","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-12T12:32:45.138412Z"},"links":{"citing_paper":"/paper/2411.17182"},"observation_digest":"sha256:ef4e001deebcdb7ee5f4a5da394c21da4d90d354f10ec5df81a9495a2b542e7a","observation_id":"3344aab9-2dbb-4b06-8a69-8fcaf690904a","resolution":{"observed_at":"2026-08-12T12:32:45.138412Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T12:32:45.143259Z","title":"Exploring generalization in deep learning","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2411.17182","last_updated":"2024-11-26T07:44:57Z","snapshot_observed_at":"2026-08-12T12:22:29.940094Z","submitted_at":"2024-11-26T07:44:57Z","title":"An In-depth Investigation of Sparse Rate Reduction in Transformer-like Models","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-12T12:32:45.143259Z"},"links":{"citing_paper":"/paper/2411.17182"},"observation_digest":"sha256:9005b8786ff1b1b4d78f7aeb0e7f5e2f053afb6c80bbd1a58b45a9a9f65e5ab0","observation_id":"0584541d-c4ff-4f99-bb6e-cf0ad3df2db9","resolution":{"observed_at":"2026-08-12T12:32:45.143259Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T12:32:45.643307Z","title":"A PAC-bayesian approach to spectrally-normalized margin bounds for neural networks","venue":null,"work_id":"dab549d5-55be-43a2-a1c0-278739c8a2be","year":2018},"citing_paper":{"arxiv_id":"2411.17182","last_updated":"2024-11-26T07:44:57Z","snapshot_observed_at":"2026-08-12T12:22:29.940094Z","submitted_at":"2024-11-26T07:44:57Z","title":"An In-depth Investigation of Sparse Rate Reduction in Transformer-like Models","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-12T12:32:45.147803Z"},"links":{"citing_paper":"/paper/2411.17182"},"observation_digest":"sha256:c8889f33577a48a49b02e381cae8a05420c22c481224deb4b32bd19c0bf3a464","observation_id":"d01d58ef-e403-45e8-bbeb-38bc23881973","resolution":{"observed_at":"2026-08-12T12:32:45.648850Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T12:32:45.152423Z","title":"Path-sgd: Path-normalized optimization in deep neural networks","venue":null,"work_id":null,"year":2015},"citing_paper":{"arxiv_id":"2411.17182","last_updated":"2024-11-26T07:44:57Z","snapshot_observed_at":"2026-08-12T12:22:29.940094Z","submitted_at":"2024-11-26T07:44:57Z","title":"An In-depth Investigation of Sparse Rate Reduction in Transformer-like Models","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-12T12:32:45.152423Z"},"links":{"citing_paper":"/paper/2411.17182"},"observation_digest":"sha256:552861f8ed18cb26bc3fae1d184604a9999c6172a7d13784618d7597d057e6bd","observation_id":"a6007c5e-c231-47cd-8547-3df21ea368f2","resolution":{"observed_at":"2026-08-12T12:32:45.152423Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T12:32:45.157053Z","title":"Norm-based capacity control in neural networks","venue":null,"work_id":null,"year":2015},"citing_paper":{"arxiv_id":"2411.17182","last_updated":"2024-11-26T07:44:57Z","snapshot_observed_at":"2026-08-12T12:22:29.940094Z","submitted_at":"2024-11-26T07:44:57Z","title":"An In-depth Investigation of Sparse Rate Reduction in Transformer-like Models","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-12T12:32:45.157053Z"},"links":{"citing_paper":"/paper/2411.17182"},"observation_digest":"sha256:7dd45280967388c61676025bff558515f2fed5b21d2ede7fc266fb55845a9ad2","observation_id":"a29f89a8-f7d3-4be4-a5aa-84cddea91a4d","resolution":{"observed_at":"2026-08-12T12:32:45.157053Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T12:32:45.607695Z","title":"Theoretical foundations of deep learning via sparse representations: A multilayer sparse model and its connection to convolutional neural networks","venue":null,"work_id":"0b3182b8-9203-4538-85a1-74d89d0711ac","year":2018},"citing_paper":{"arxiv_id":"2411.17182","last_updated":"2024-11-26T07:44:57Z","snapshot_observed_at":"2026-08-12T12:22:29.940094Z","submitted_at":"2024-11-26T07:44:57Z","title":"An In-depth Investigation of Sparse Rate Reduction in Transformer-like Models","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-12T12:32:45.161472Z"},"links":{"citing_paper":"/paper/2411.17182"},"observation_digest":"sha256:9977128f62f84ddbefc9d99b0f14183ba2c822e8455070488cce8cc9fa09d6b0","observation_id":"bf19aabb-2092-4d15-a8ac-824a19b2ef4d","resolution":{"observed_at":"2026-08-12T12:32:45.612840Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2201.02177","last_updated":"2022-01-06T18:43:37Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2022-01-06T18:43:37Z","title":"Grokking: Generalization Beyond Overfitting on Small Algorithmic Datasets","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2201.02177","snapshot_observed_at":"2026-08-12T12:32:45.165818Z","title":"Grokking: Gen- eralization beyond overfitting on small algorithmic datasets","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2411.17182","last_updated":"2024-11-26T07:44:57Z","snapshot_observed_at":"2026-08-12T12:22:29.940094Z","submitted_at":"2024-11-26T07:44:57Z","title":"An In-depth Investigation of Sparse Rate Reduction in Transformer-like Models","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-12T12:32:45.165818Z"},"links":{"cited_paper":"/paper/2201.02177","citing_paper":"/paper/2411.17182"},"observation_digest":"sha256:66798a0d89439ae72ec0d74d1d23d5bdd066e08540ced62c21eecc059060ffd5","observation_id":"68370c3f-74ac-4fc1-865f-be6b2b4cf1b2","resolution":{"observed_at":"2026-08-12T12:32:45.165818Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T12:32:45.592323Z","title":"Hopfield networks is all you need","venue":null,"work_id":"ce476448-5c8e-4484-bed0-2a64d8996828","year":2021},"citing_paper":{"arxiv_id":"2411.17182","last_updated":"2024-11-26T07:44:57Z","snapshot_observed_at":"2026-08-12T12:22:29.940094Z","submitted_at":"2024-11-26T07:44:57Z","title":"An In-depth Investigation of Sparse Rate Reduction in Transformer-like Models","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-12T12:32:45.170405Z"},"links":{"citing_paper":"/paper/2411.17182"},"observation_digest":"sha256:18b319386310e2ee1d17a0710e40c0f9f92141c3e32d7183752431e23a938963","observation_id":"9c16a147-3160-4d09-925c-5c70977a2500","resolution":{"observed_at":"2026-08-12T12:32:45.596998Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T12:32:45.576735Z","title":"Unraveling attention via convex duality: Analysis and interpretations of vision transformers","venue":null,"work_id":"6cdf96af-5774-4d50-962d-3ab45b3a766b","year":2022},"citing_paper":{"arxiv_id":"2411.17182","last_updated":"2024-11-26T07:44:57Z","snapshot_observed_at":"2026-08-12T12:22:29.940094Z","submitted_at":"2024-11-26T07:44:57Z","title":"An In-depth Investigation of Sparse Rate Reduction in Transformer-like Models","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-12T12:32:45.174964Z"},"links":{"citing_paper":"/paper/2411.17182"},"observation_digest":"sha256:2361e46ea04416242b8d729f572ce6f7c3683bea1274efbc539a445477a28b83","observation_id":"d047be21-d60e-4e4d-96ac-a3cb81671714","resolution":{"observed_at":"2026-08-12T12:32:45.582018Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T12:32:45.561479Z","title":"Biological learning in key-value memory networks","venue":null,"work_id":"fe44f3fd-e744-42e7-8192-b48db987b2d3","year":2021},"citing_paper":{"arxiv_id":"2411.17182","last_updated":"2024-11-26T07:44:57Z","snapshot_observed_at":"2026-08-12T12:22:29.940094Z","submitted_at":"2024-11-26T07:44:57Z","title":"An In-depth Investigation of Sparse Rate Reduction in Transformer-like Models","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-12T12:32:45.179474Z"},"links":{"citing_paper":"/paper/2411.17182"},"observation_digest":"sha256:6edc1ad3f354a2a115d4ffc6e74c7cd0e0161ca75bfe42e08b7cc4d2476327d8","observation_id":"e8f5ac9e-88a5-44d7-a4e7-f40c3afbcb3f","resolution":{"observed_at":"2026-08-12T12:32:45.566234Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T12:32:45.183774Z","title":"On the uniform convergence of relative frequencies of events to their probabilities","venue":null,"work_id":null,"year":2015},"citing_paper":{"arxiv_id":"2411.17182","last_updated":"2024-11-26T07:44:57Z","snapshot_observed_at":"2026-08-12T12:22:29.940094Z","submitted_at":"2024-11-26T07:44:57Z","title":"An In-depth Investigation of Sparse Rate Reduction in Transformer-like Models","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-12T12:32:45.183774Z"},"links":{"citing_paper":"/paper/2411.17182"},"observation_digest":"sha256:bbacf13554d7f51e438c8922362bd8b5f1f4eebdefa1700ad226fa1bd8d55ab2","observation_id":"7ceec1d0-12e0-4188-9409-b85ef544dc90","resolution":{"observed_at":"2026-08-12T12:32:45.183774Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T12:32:45.188711Z","title":"Attention is all you need","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2411.17182","last_updated":"2024-11-26T07:44:57Z","snapshot_observed_at":"2026-08-12T12:22:29.940094Z","submitted_at":"2024-11-26T07:44:57Z","title":"An In-depth Investigation of Sparse Rate Reduction in Transformer-like Models","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-12T12:32:45.188711Z"},"links":{"citing_paper":"/paper/2411.17182"},"observation_digest":"sha256:c3586bcf57d934d631680d43544938581361a57c8fa80684b2399ad7aede4ac6","observation_id":"45595754-205e-4650-8973-eb26e14d9a7c","resolution":{"observed_at":"2026-08-12T12:32:45.188711Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T12:32:45.193362Z","title":"Interpretability in the wild: a circuit for indirect object identification in GPT-2 small","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2411.17182","last_updated":"2024-11-26T07:44:57Z","snapshot_observed_at":"2026-08-12T12:22:29.940094Z","submitted_at":"2024-11-26T07:44:57Z","title":"An In-depth Investigation of Sparse Rate Reduction in Transformer-like Models","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-12T12:32:45.193362Z"},"links":{"citing_paper":"/paper/2411.17182"},"observation_digest":"sha256:36bdcb29ccb8709767cb6dad2bb666bb8323cd32c877a767cb27e8e9a2015258","observation_id":"54fbf4f8-9de5-4311-930a-8de8af52ca77","resolution":{"observed_at":"2026-08-12T12:32:45.193362Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T12:32:45.517104Z","title":"Thinking like transformers","venue":null,"work_id":"b511363e-2548-485d-954d-a9e58d771cee","year":2021},"citing_paper":{"arxiv_id":"2411.17182","last_updated":"2024-11-26T07:44:57Z","snapshot_observed_at":"2026-08-12T12:22:29.940094Z","submitted_at":"2024-11-26T07:44:57Z","title":"An In-depth Investigation of Sparse Rate Reduction in Transformer-like Models","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-12T12:32:45.198222Z"},"links":{"citing_paper":"/paper/2411.17182"},"observation_digest":"sha256:74727e01fdd4f83fefcb0ae411226ac474f64ec59b436432f00fad3ea784c039","observation_id":"16708be7-7ec9-4462-b6f0-49fd059e8c47","resolution":{"observed_at":"2026-08-12T12:32:45.522435Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T12:32:45.501246Z","title":"Graph neural networks inspired by classical iterative algorithms","venue":null,"work_id":"f09495bf-c1be-42e2-8044-939e027dfa0a","year":2021},"citing_paper":{"arxiv_id":"2411.17182","last_updated":"2024-11-26T07:44:57Z","snapshot_observed_at":"2026-08-12T12:22:29.940094Z","submitted_at":"2024-11-26T07:44:57Z","title":"An In-depth Investigation of Sparse Rate Reduction in Transformer-like Models","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-12T12:32:45.202827Z"},"links":{"citing_paper":"/paper/2411.17182"},"observation_digest":"sha256:bd80818d64fa04b3abad32f7ce29420ec645edf0ff67922c361b4156933a8659","observation_id":"28500a26-2d01-4722-86b7-8319da737fa4","resolution":{"observed_at":"2026-08-12T12:32:45.505906Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T12:32:45.486197Z","title":"Transformers from an optimization perspective","venue":null,"work_id":"e2d468e9-8632-4d11-90f3-5407e6b98b9c","year":2022},"citing_paper":{"arxiv_id":"2411.17182","last_updated":"2024-11-26T07:44:57Z","snapshot_observed_at":"2026-08-12T12:22:29.940094Z","submitted_at":"2024-11-26T07:44:57Z","title":"An In-depth Investigation of Sparse Rate Reduction in Transformer-like Models","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-12T12:32:45.207058Z"},"links":{"citing_paper":"/paper/2411.17182"},"observation_digest":"sha256:1b5403b1584f7da187fc0cc7a7fdcc84f1bd72802aa34d39eac27e7f73ff3be8","observation_id":"5819e8a2-a4ec-4038-be03-839a1c45ad48","resolution":{"observed_at":"2026-08-12T12:32:45.490925Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T12:32:45.470719Z","title":"Attentionviz: A global view of transformer attention","venue":null,"work_id":"f875ec19-fd8f-43c2-9b9f-9fa201fd1ea6","year":2023},"citing_paper":{"arxiv_id":"2411.17182","last_updated":"2024-11-26T07:44:57Z","snapshot_observed_at":"2026-08-12T12:22:29.940094Z","submitted_at":"2024-11-26T07:44:57Z","title":"An In-depth Investigation of Sparse Rate Reduction in Transformer-like Models","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-12T12:32:45.211699Z"},"links":{"citing_paper":"/paper/2411.17182"},"observation_digest":"sha256:283bef8c616670d43b2b3f6c9be2e7299337172299134cf89803913b3b2fd188","observation_id":"0b214848-862c-4a0d-8529-e072e6f554e2","resolution":{"observed_at":"2026-08-12T12:32:45.475648Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T12:32:45.455686Z","title":"White-box transformers via sparse rate reduction","venue":null,"work_id":"4486a08b-971a-4437-9e84-7daddee18b10","year":2023},"citing_paper":{"arxiv_id":"2411.17182","last_updated":"2024-11-26T07:44:57Z","snapshot_observed_at":"2026-08-12T12:22:29.940094Z","submitted_at":"2024-11-26T07:44:57Z","title":"An In-depth Investigation of Sparse Rate Reduction in Transformer-like Models","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-12T12:32:45.216547Z"},"links":{"citing_paper":"/paper/2411.17182"},"observation_digest":"sha256:2e6b3e8f5b98f5ff435e49394df89d34de3a3710b85acdbfe08aa26493b55e05","observation_id":"12cf0273-156d-4d0c-b1cc-529692a0833f","resolution":{"observed_at":"2026-08-12T12:32:45.460496Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T12:32:45.438391Z","title":"Learning diverse and discriminative representations via the principle of maximal coding rate reduction","venue":null,"work_id":"db65bebc-d782-4629-a3c9-d53b2e9407ef","year":2020},"citing_paper":{"arxiv_id":"2411.17182","last_updated":"2024-11-26T07:44:57Z","snapshot_observed_at":"2026-08-12T12:22:29.940094Z","submitted_at":"2024-11-26T07:44:57Z","title":"An In-depth Investigation of Sparse Rate Reduction in Transformer-like Models","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-12T12:32:45.221157Z"},"links":{"citing_paper":"/paper/2411.17182"},"observation_digest":"sha256:7136c7ce06b7f93368f55867b0c596c1d5948f9eec224c6d82c59b3110ceab66","observation_id":"048031ec-59eb-4fc9-a8a0-d30640be511d","resolution":{"observed_at":"2026-08-12T12:32:45.445097Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2206.04301","last_updated":"2023-02-17T23:18:30Z","snapshot_observed_at":"2026-07-06T13:18:57.269780Z","submitted_at":"2022-06-09T06:30:17Z","title":"Unveiling Transformers with LEGO: a synthetic reasoning task","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2206.04301","snapshot_observed_at":"2026-08-12T12:32:45.226075Z","title":"Unveiling transformers with lego: a synthetic reasoning task","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2411.17182","last_updated":"2024-11-26T07:44:57Z","snapshot_observed_at":"2026-08-12T12:22:29.940094Z","submitted_at":"2024-11-26T07:44:57Z","title":"An In-depth Investigation of Sparse Rate Reduction in Transformer-like Models","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-12T12:32:45.226075Z"},"links":{"cited_paper":"/paper/2206.04301","citing_paper":"/paper/2411.17182"},"observation_digest":"sha256:5655caa7c6d717be27e4e7eb25569d3d39785c16425c1280e59ccb6d4edb7b47","observation_id":"af831bc5-e02c-4fd1-8cad-c1d2357de761","resolution":{"observed_at":"2026-08-12T12:32:45.226075Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T12:32:45.061926Z","title":null,"venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2411.17182","last_updated":"2024-11-26T07:44:57Z","snapshot_observed_at":"2026-08-12T12:22:29.940094Z","submitted_at":"2024-11-26T07:44:57Z","title":"An In-depth Investigation of Sparse Rate Reduction in Transformer-like Models","version":1},"reference_index":2021,"source":"pdf_text","source_observed_at":"2026-08-12T12:32:45.061926Z"},"links":{"citing_paper":"/paper/2411.17182"},"observation_digest":"sha256:d85ca9f6ff7915b7d77f504b0f49a92e9734faf1263a205b0fe4ba0c87d4b74a","observation_id":"b7346d03-3e0f-4e9f-9460-f178836751b5","resolution":{"observed_at":"2026-08-12T12:32:45.061926Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2411.17182","last_updated":"2024-11-26T07:44:57Z","latest_version":1,"primary_category":"cs.LG","snapshot_observed_at":"2026-08-12T12:22:29.940094Z","submitted_at":"2024-11-26T07:44:57Z","title":"An In-depth Investigation of Sparse Rate Reduction in Transformer-like Models"},"reference_resolution":{"displayed":48,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":21,"verified_exact":0,"verified_fuzzy":27},"total_outbound_references":48},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"thesis":"As of 13 August 2026, this Paper Citation Record lists 48 of 48 outbound references and 0 inbound Pith citation observations for arXiv:2411.17182."}