{"as_of":"2026-08-08T18:55:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:121798ccc63f014af29303193e6a409f4ef3afff85457b822dc998307e7c9bd5","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":15,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":15,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-08T06:32:00.761636+00:00","state":"measured"},{"denominator":15,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":15,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T04:59:24.336490Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":1,"source":"arxiv_reference","source_observed_at":"2026-08-05T02:28:24.338817Z","state":"measured"}],"external_citation_measurements":[{"count":7,"observed_at":"2026-08-05T02:28:24.338817Z","source":"arxiv_reference"}],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2310.16028","last_updated":"2023-10-24T17:43:29Z","snapshot_observed_at":"2026-07-06T16:37:58.121261Z","submitted_at":"2023-10-24T17:43:29Z","title":"What Algorithms can Transformers Learn? A Study in Length Generalization","version":1},"cited_work":{"arxiv_id":"2310.16028","doi":"10.48550/arxiv.2310.16028","metadata_source":"arxiv_reference","pith_arxiv_id":"2310.16028","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"What algorithms can transformers learn? a study in length generalization","venue":"arXiv (Cornell University)","work_id":"f5a31c3d-1b42-4138-86e3-a61bc8c7c3a6","year":2024},"citing_paper":{"arxiv_id":"2403.07974","last_updated":"2024-06-06T17:41:21Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-03-12T17:58:04Z","title":"LiveCodeBench: Holistic and Contamination Free Evaluation of Large Language Models for Code","version":2},"reference_index":94,"source":"arxiv_source","source_observed_at":"2026-05-10T17:34:42.565806Z"},"links":{"cited_paper":"/paper/2310.16028","citing_paper":"/paper/2403.07974"},"observation_digest":"sha256:8b09f7c1ea1a46cb51301674faf2c5af6ce4826597e606b7e0ed6f8c17f84c87","observation_id":"eb08c33d-41a8-4903-af3d-095ee9979da4","resolution":{"observed_at":"2026-05-10T17:34:43.087593Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.16028","last_updated":"2023-10-24T17:43:29Z","snapshot_observed_at":"2026-07-06T16:37:58.121261Z","submitted_at":"2023-10-24T17:43:29Z","title":"What Algorithms can Transformers Learn? A Study in Length Generalization","version":1},"cited_work":{"arxiv_id":"2310.16028","doi":"10.48550/arxiv.2310.16028","metadata_source":"arxiv_reference","pith_arxiv_id":"2310.16028","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"What algorithms can transformers learn? a study in length generalization","venue":"arXiv (Cornell University)","work_id":"f5a31c3d-1b42-4138-86e3-a61bc8c7c3a6","year":2024},"citing_paper":{"arxiv_id":"2502.09741","last_updated":"2026-04-21T01:47:04Z","snapshot_observed_at":"2026-07-31T13:50:10.330381Z","submitted_at":"2025-02-13T19:54:59Z","title":"FoNE: Precise Single-Token Number Embeddings via Fourier Features","version":2},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-05-23T03:07:37.363965Z"},"links":{"cited_paper":"/paper/2310.16028","citing_paper":"/paper/2502.09741"},"observation_digest":"sha256:39bd142aaae696759b3a0ca58dfe58aac8db8296e4f19418103179d6c3ee95d3","observation_id":"76c715e3-56a9-41b9-9470-18d9866511e1","resolution":{"observed_at":"2026-05-23T03:12:28.730236Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.16028","last_updated":"2023-10-24T17:43:29Z","snapshot_observed_at":"2026-07-06T16:37:58.121261Z","submitted_at":"2023-10-24T17:43:29Z","title":"What Algorithms can Transformers Learn? A Study in Length Generalization","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.16028","snapshot_observed_at":"2026-08-07T04:59:24.336490Z","title":"What algorithms can transformers learn? a study in length generalization","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.09251","last_updated":"2025-08-04T16:57:32Z","snapshot_observed_at":"2026-08-07T04:50:54.681994Z","submitted_at":"2025-06-10T21:22:51Z","title":"Extrapolation by Association: Length Generalization Transfer in Transformers","version":2},"reference_index":46,"source":"arxiv_source","source_observed_at":"2026-08-07T04:59:24.336490Z"},"links":{"cited_paper":"/paper/2310.16028","citing_paper":"/paper/2506.09251"},"observation_digest":"sha256:04680cfda77be038619388aac6d16a25e3a6afd08661042fa82885cb9465de7f","observation_id":"1b4e5d1f-9277-4d54-a717-643538901f11","resolution":{"observed_at":"2026-08-07T04:59:24.336490Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.16028","last_updated":"2023-10-24T17:43:29Z","snapshot_observed_at":"2026-07-06T16:37:58.121261Z","submitted_at":"2023-10-24T17:43:29Z","title":"What Algorithms can Transformers Learn? A Study in Length Generalization","version":1},"cited_work":{"arxiv_id":"2310.16028","doi":"10.48550/arxiv.2310.16028","metadata_source":"arxiv_reference","pith_arxiv_id":"2310.16028","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"What algorithms can transformers learn? a study in length generalization","venue":"arXiv (Cornell University)","work_id":"f5a31c3d-1b42-4138-86e3-a61bc8c7c3a6","year":2024},"citing_paper":{"arxiv_id":"2507.12549","last_updated":"2026-04-28T23:40:54Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-16T18:01:26Z","title":"The Serial Scaling Hypothesis","version":4},"reference_index":137,"source":"arxiv_source","source_observed_at":"2026-05-19T04:08:11.344622Z"},"links":{"cited_paper":"/paper/2310.16028","citing_paper":"/paper/2507.12549"},"observation_digest":"sha256:b8e098819333b89292784b5f940f99f8568b875d53bb09c33d72e5faa30dcc20","observation_id":"e642d740-59e9-406b-b362-41984e948cee","resolution":{"observed_at":"2026-05-19T04:12:02.401992Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.16028","last_updated":"2023-10-24T17:43:29Z","snapshot_observed_at":"2026-07-06T16:37:58.121261Z","submitted_at":"2023-10-24T17:43:29Z","title":"What Algorithms can Transformers Learn? A Study in Length Generalization","version":1},"cited_work":{"arxiv_id":"2310.16028","doi":"10.48550/arxiv.2310.16028","metadata_source":"arxiv_reference","pith_arxiv_id":"2310.16028","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"What algorithms can transformers learn? a study in length generalization","venue":"arXiv (Cornell University)","work_id":"f5a31c3d-1b42-4138-86e3-a61bc8c7c3a6","year":2024},"citing_paper":{"arxiv_id":"2602.01651","last_updated":"2026-04-21T07:48:12Z","snapshot_observed_at":"2026-07-06T22:44:04.951815Z","submitted_at":"2026-02-02T05:11:48Z","title":"On the Spatiotemporal Dynamics of Generalization in Neural Networks","version":2},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-16T08:59:44.016444Z"},"links":{"cited_paper":"/paper/2310.16028","citing_paper":"/paper/2602.01651"},"observation_digest":"sha256:890bfc286c3a06f0039a7eca1a2ad3605e00ff157aec7b78e6139d2905a429d3","observation_id":"67b93f2e-92fa-4641-81ee-4c39d0d58151","resolution":{"observed_at":"2026-05-16T09:00:46.836643Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.16028","last_updated":"2023-10-24T17:43:29Z","snapshot_observed_at":"2026-07-06T16:37:58.121261Z","submitted_at":"2023-10-24T17:43:29Z","title":"What Algorithms can Transformers Learn? A Study in Length Generalization","version":1},"cited_work":{"arxiv_id":"2310.16028","doi":"10.48550/arxiv.2310.16028","metadata_source":"arxiv_reference","pith_arxiv_id":"2310.16028","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"What algorithms can transformers learn? a study in length generalization","venue":"arXiv (Cornell University)","work_id":"f5a31c3d-1b42-4138-86e3-a61bc8c7c3a6","year":2024},"citing_paper":{"arxiv_id":"2603.06870","last_updated":"2026-04-22T15:40:03Z","snapshot_observed_at":"2026-07-06T22:48:13.053749Z","submitted_at":"2026-03-06T20:42:41Z","title":"LEAD: Breaking the No-Recovery Bottleneck in Long-Horizon Reasoning","version":2},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-05-15T14:34:14.413131Z"},"links":{"cited_paper":"/paper/2310.16028","citing_paper":"/paper/2603.06870"},"observation_digest":"sha256:f790f8084ba73e449a8e4c272687a4d8bff170b65b8d692d455dcf3cf750f775","observation_id":"5701e019-3466-4348-95de-db0394411527","resolution":{"observed_at":"2026-05-15T14:35:55.714967Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.16028","last_updated":"2023-10-24T17:43:29Z","snapshot_observed_at":"2026-07-06T16:37:58.121261Z","submitted_at":"2023-10-24T17:43:29Z","title":"What Algorithms can Transformers Learn? A Study in Length Generalization","version":1},"cited_work":{"arxiv_id":"2310.16028","doi":"10.48550/arxiv.2310.16028","metadata_source":"arxiv_reference","pith_arxiv_id":"2310.16028","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"What algorithms can transformers learn? a study in length generalization","venue":"arXiv (Cornell University)","work_id":"f5a31c3d-1b42-4138-86e3-a61bc8c7c3a6","year":2024},"citing_paper":{"arxiv_id":"2603.29069","last_updated":"2026-04-05T18:40:06Z","snapshot_observed_at":"2026-08-07T17:56:32.986328Z","submitted_at":"2026-03-30T23:15:21Z","title":"On the Mirage of Long-Range Dependency, with an Application to Integer Multiplication","version":2},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-05-14T21:07:37.032208Z"},"links":{"cited_paper":"/paper/2310.16028","citing_paper":"/paper/2603.29069"},"observation_digest":"sha256:61aa40630ce907cc6630dfd84cb1e867cc69b9434f88c40cff84ab04d7cb15d4","observation_id":"b2425abd-d1c5-4029-b52c-0732cb79c3d1","resolution":{"observed_at":"2026-05-14T21:07:57.577421Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.16028","last_updated":"2023-10-24T17:43:29Z","snapshot_observed_at":"2026-07-06T16:37:58.121261Z","submitted_at":"2023-10-24T17:43:29Z","title":"What Algorithms can Transformers Learn? A Study in Length Generalization","version":1},"cited_work":{"arxiv_id":"2310.16028","doi":"10.48550/arxiv.2310.16028","metadata_source":"arxiv_reference","pith_arxiv_id":"2310.16028","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"What algorithms can transformers learn? a study in length generalization","venue":"arXiv (Cornell University)","work_id":"f5a31c3d-1b42-4138-86e3-a61bc8c7c3a6","year":2024},"citing_paper":{"arxiv_id":"2604.15306","last_updated":"2026-04-16T17:59:43Z","snapshot_observed_at":"2026-07-06T23:02:53.677687Z","submitted_at":"2026-04-16T17:59:43Z","title":"Generalization in LLM Problem Solving: The Case of the Shortest Path","version":1},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-05-10T10:37:45.355872Z"},"links":{"cited_paper":"/paper/2310.16028","citing_paper":"/paper/2604.15306"},"observation_digest":"sha256:9a66bd8a9d16c5478773ee3127c21244eb79cfb0e46322befc5ccb48749e12e7","observation_id":"af397df4-be69-411a-9e20-43cbdecd80e1","resolution":{"observed_at":"2026-05-10T10:39:38.085516Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.16028","last_updated":"2023-10-24T17:43:29Z","snapshot_observed_at":"2026-07-06T16:37:58.121261Z","submitted_at":"2023-10-24T17:43:29Z","title":"What Algorithms can Transformers Learn? A Study in Length Generalization","version":1},"cited_work":{"arxiv_id":"2310.16028","doi":"10.48550/arxiv.2310.16028","metadata_source":"arxiv_reference","pith_arxiv_id":"2310.16028","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"What algorithms can transformers learn? a study in length generalization","venue":"arXiv (Cornell University)","work_id":"f5a31c3d-1b42-4138-86e3-a61bc8c7c3a6","year":2024},"citing_paper":{"arxiv_id":"2604.17857","last_updated":"2026-04-21T13:11:34Z","snapshot_observed_at":"2026-07-06T23:04:50.935335Z","submitted_at":"2026-04-20T06:10:50Z","title":"On the Emergence of Syntax by Means of Local Interaction","version":2},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-05-10T04:16:41.473079Z"},"links":{"cited_paper":"/paper/2310.16028","citing_paper":"/paper/2604.17857"},"observation_digest":"sha256:c3b596c8545134e229c48c870fe8624001009b27339c609691dff0b617ab47b3","observation_id":"a75bf0e2-a315-4ede-8040-3197835301c0","resolution":{"observed_at":"2026-05-10T04:20:03.801289Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.16028","last_updated":"2023-10-24T17:43:29Z","snapshot_observed_at":"2026-07-06T16:37:58.121261Z","submitted_at":"2023-10-24T17:43:29Z","title":"What Algorithms can Transformers Learn? A Study in Length Generalization","version":1},"cited_work":{"arxiv_id":"2310.16028","doi":"10.48550/arxiv.2310.16028","metadata_source":"arxiv_reference","pith_arxiv_id":"2310.16028","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"What algorithms can transformers learn? a study in length generalization","venue":"arXiv (Cornell University)","work_id":"f5a31c3d-1b42-4138-86e3-a61bc8c7c3a6","year":2024},"citing_paper":{"arxiv_id":"2604.25166","last_updated":"2026-04-28T03:15:44Z","snapshot_observed_at":"2026-07-06T23:11:02.043135Z","submitted_at":"2026-04-28T03:15:44Z","title":"Training Transformers as a Universal Computer","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-05-07T16:36:19.729400Z"},"links":{"cited_paper":"/paper/2310.16028","citing_paper":"/paper/2604.25166"},"observation_digest":"sha256:a298d65bf293154bb752e797d1605a0835210138727c8a18bbfdf7c39c738b16","observation_id":"714621d3-602d-4d81-8078-ef9fc8d331d0","resolution":{"observed_at":"2026-05-11T23:36:38.589963Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.16028","last_updated":"2023-10-24T17:43:29Z","snapshot_observed_at":"2026-07-06T16:37:58.121261Z","submitted_at":"2023-10-24T17:43:29Z","title":"What Algorithms can Transformers Learn? A Study in Length Generalization","version":1},"cited_work":{"arxiv_id":"2310.16028","doi":"10.48550/arxiv.2310.16028","metadata_source":"arxiv_reference","pith_arxiv_id":"2310.16028","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"What algorithms can transformers learn? a study in length generalization","venue":"arXiv (Cornell University)","work_id":"f5a31c3d-1b42-4138-86e3-a61bc8c7c3a6","year":2024},"citing_paper":{"arxiv_id":"2606.00183","last_updated":"2026-05-29T14:58:03Z","snapshot_observed_at":"2026-07-06T23:40:56.510371Z","submitted_at":"2026-05-29T14:58:03Z","title":"Agentic Transformers Provably Learn to Search via Reinforcement Learning","version":1},"reference_index":33,"source":"arxiv_source","source_observed_at":"2026-06-28T23:26:28.158991Z"},"links":{"cited_paper":"/paper/2310.16028","citing_paper":"/paper/2606.00183"},"observation_digest":"sha256:7b6d3e26760e06cf2a169050b9f879b804172780f8b3cdc6922fb65627fca441","observation_id":"43ee4f3f-1d48-407f-b1e8-2fa11c7325b9","resolution":{"observed_at":"2026-06-28T23:42:49.913561Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.16028","last_updated":"2023-10-24T17:43:29Z","snapshot_observed_at":"2026-07-06T16:37:58.121261Z","submitted_at":"2023-10-24T17:43:29Z","title":"What Algorithms can Transformers Learn? A Study in Length Generalization","version":1},"cited_work":{"arxiv_id":"2310.16028","doi":"10.48550/arxiv.2310.16028","metadata_source":"arxiv_reference","pith_arxiv_id":"2310.16028","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"What algorithms can transformers learn? a study in length generalization","venue":"arXiv (Cornell University)","work_id":"f5a31c3d-1b42-4138-86e3-a61bc8c7c3a6","year":2024},"citing_paper":{"arxiv_id":"2606.21884","last_updated":"2026-06-20T04:52:14Z","snapshot_observed_at":"2026-08-03T01:34:19.883808Z","submitted_at":"2026-06-20T04:52:14Z","title":"A Verifiable Search Is Not a Learnable Chain-of-Thought","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-06-26T12:35:20.698118Z"},"links":{"cited_paper":"/paper/2310.16028","citing_paper":"/paper/2606.21884"},"observation_digest":"sha256:8f03959a19ab5a7475f55a74a1688ca8129df70b0a0f1f202b2743f5e8429ebb","observation_id":"775b5221-9025-4310-a56f-9c60ee1ab94a","resolution":{"observed_at":"2026-07-04T07:49:39.727613Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.16028","last_updated":"2023-10-24T17:43:29Z","snapshot_observed_at":"2026-07-06T16:37:58.121261Z","submitted_at":"2023-10-24T17:43:29Z","title":"What Algorithms can Transformers Learn? A Study in Length Generalization","version":1},"cited_work":{"arxiv_id":"2310.16028","doi":"10.48550/arxiv.2310.16028","metadata_source":"arxiv_reference","pith_arxiv_id":"2310.16028","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"What algorithms can transformers learn? a study in length generalization","venue":"arXiv (Cornell University)","work_id":"f5a31c3d-1b42-4138-86e3-a61bc8c7c3a6","year":2024},"citing_paper":{"arxiv_id":"2606.29983","last_updated":"2026-06-29T08:58:09Z","snapshot_observed_at":"2026-08-03T21:27:41.768539Z","submitted_at":"2026-06-29T08:58:09Z","title":"Stabilizing Extrapolation in Looped Transformers via Learned Stochastic Stopping","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-06-30T07:29:23.786653Z"},"links":{"cited_paper":"/paper/2310.16028","citing_paper":"/paper/2606.29983"},"observation_digest":"sha256:34a90c79b78461d7606264d0815df70b822b7ad7461abf400a5244812a56e1af","observation_id":"7e817d48-85e3-476a-9915-0887468e949d","resolution":{"observed_at":"2026-06-30T07:34:21.673836Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.16028","last_updated":"2023-10-24T17:43:29Z","snapshot_observed_at":"2026-07-06T16:37:58.121261Z","submitted_at":"2023-10-24T17:43:29Z","title":"What Algorithms can Transformers Learn? A Study in Length Generalization","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.16028","snapshot_observed_at":"2026-07-14T03:23:31.727605Z","title":"arXiv preprint arXiv:2310.16028 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.11760","last_updated":"2026-07-13T16:22:36Z","snapshot_observed_at":"2026-08-06T22:36:16.299904Z","submitted_at":"2026-07-13T16:22:36Z","title":"From Expressivity to Sample Complexity: Narrow Teachers for Transformers via C-RASP","version":1},"reference_index":6,"source":"arxiv_source","source_observed_at":"2026-07-14T03:23:31.727605Z"},"links":{"cited_paper":"/paper/2310.16028","citing_paper":"/paper/2607.11760"},"observation_digest":"sha256:935c493309dfb30d24f505f8d7bbca60bb73eef869ec3e1f846121ba434a9dea","observation_id":"26a39aa7-3826-4c2e-95b0-638661fee8ca","resolution":{"observed_at":"2026-07-14T03:23:31.727605Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.16028","last_updated":"2023-10-24T17:43:29Z","snapshot_observed_at":"2026-07-06T16:37:58.121261Z","submitted_at":"2023-10-24T17:43:29Z","title":"What Algorithms can Transformers Learn? A Study in Length Generalization","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.16028","snapshot_observed_at":"2026-08-01T17:31:50.953124Z","title":"What algorithms can transformers learn? a study in length generalization","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.17624","last_updated":"2026-07-20T07:26:17Z","snapshot_observed_at":"2026-08-07T13:43:21.717264Z","submitted_at":"2026-07-20T07:26:17Z","title":"Can Transformers Really Do It All? On the Compatibility of Inductive Biases Across Tasks","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-01T17:31:50.953124Z"},"links":{"cited_paper":"/paper/2310.16028","citing_paper":"/paper/2607.17624"},"observation_digest":"sha256:8bdbafbd18c3b808bfece7d59877b962b56bfb637f4e920815ff3cd6eb60351a","observation_id":"f90c2cbe-7c3d-468e-8328-59cc4cc74438","resolution":{"observed_at":"2026-08-01T17:31:50.953124Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2310.16028/citation-record","integrity":"/paper/2310.16028/integrity","json":"/paper/2310.16028/citation-record.json","paper":"/paper/2310.16028"},"outbound":[],"paper":{"arxiv_id":"2310.16028","last_updated":"2023-10-24T17:43:29Z","latest_version":1,"primary_category":"cs.LG","snapshot_observed_at":"2026-07-06T16:37:58.121261Z","submitted_at":"2023-10-24T17:43:29Z","title":"What Algorithms can Transformers Learn? A Study in Length Generalization"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"thesis":"As of 8 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 15 inbound Pith citation observations for arXiv:2310.16028."}