{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:BRGD5OASOPAGUWAJAVEBFOQ52X","short_pith_number":"pith:BRGD5OAS","schema_version":"1.0","canonical_sha256":"0c4c3eb81273c06a5809054812ba1dd5e2dd65fa34d61fe936503ab3f544c637","source":{"kind":"arxiv","id":"2310.16028","version":1},"attestation_state":"computed","paper":{"title":"What Algorithms can Transformers Learn? A Study in Length Generalization","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL","stat.ML"],"primary_cat":"cs.LG","authors_text":"Arwen Bradley, Etai Littwin, Hattie Zhou, Josh Susskind, Noam Razin, Omid Saremi, Preetum Nakkiran, Samy Bengio","submitted_at":"2023-10-24T17:43:29Z","abstract_excerpt":"Large language models exhibit surprising emergent generalization properties, yet also struggle on many simple reasoning tasks such as arithmetic and parity. This raises the question of if and when Transformer models can learn the true algorithm for solving a task. We study the scope of Transformers' abilities in the specific setting of length generalization on algorithmic tasks. Here, we propose a unifying framework to understand when and how Transformers can exhibit strong length generalization on a given task. Specifically, we leverage RASP (Weiss et al., 2021) -- a programming language desi"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2310.16028","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2023-10-24T17:43:29Z","cross_cats_sorted":["cs.AI","cs.CL","stat.ML"],"title_canon_sha256":"3723b60a77a7ba3d7830c537583b740330cf519531a5a06abb45d43d896ab75a","abstract_canon_sha256":"a65f630f9d8a5c6e5e4344dbfe6ca016fa4b54a9e0613f4ef415fce542e2ede2"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:04:33.999205Z","signature_b64":"XI88j5DF/gAIr9DhlBzGADued9dgpUxNKTmVi4I1ka7WWtBD9XMDMkDNXEU2p3h+9/Rt9l5DGJyZSAnjOAhZDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"0c4c3eb81273c06a5809054812ba1dd5e2dd65fa34d61fe936503ab3f544c637","last_reissued_at":"2026-07-05T07:04:33.998741Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:04:33.998741Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"What Algorithms can Transformers Learn? A Study in Length Generalization","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL","stat.ML"],"primary_cat":"cs.LG","authors_text":"Arwen Bradley, Etai Littwin, Hattie Zhou, Josh Susskind, Noam Razin, Omid Saremi, Preetum Nakkiran, Samy Bengio","submitted_at":"2023-10-24T17:43:29Z","abstract_excerpt":"Large language models exhibit surprising emergent generalization properties, yet also struggle on many simple reasoning tasks such as arithmetic and parity. This raises the question of if and when Transformer models can learn the true algorithm for solving a task. We study the scope of Transformers' abilities in the specific setting of length generalization on algorithmic tasks. Here, we propose a unifying framework to understand when and how Transformers can exhibit strong length generalization on a given task. Specifically, we leverage RASP (Weiss et al., 2021) -- a programming language desi"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2310.16028","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2310.16028/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2310.16028","created_at":"2026-07-05T07:04:33.998796+00:00"},{"alias_kind":"arxiv_version","alias_value":"2310.16028v1","created_at":"2026-07-05T07:04:33.998796+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2310.16028","created_at":"2026-07-05T07:04:33.998796+00:00"},{"alias_kind":"pith_short_12","alias_value":"BRGD5OASOPAG","created_at":"2026-07-05T07:04:33.998796+00:00"},{"alias_kind":"pith_short_16","alias_value":"BRGD5OASOPAGUWAJ","created_at":"2026-07-05T07:04:33.998796+00:00"},{"alias_kind":"pith_short_8","alias_value":"BRGD5OAS","created_at":"2026-07-05T07:04:33.998796+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":12,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.21884","citing_title":"A Verifiable Search Is Not a Learnable Chain-of-Thought","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2606.29983","citing_title":"Stabilizing Extrapolation in Looped Transformers via Learned Stochastic Stopping","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2606.00183","citing_title":"Agentic Transformers Provably Learn to Search via Reinforcement Learning","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2502.09741","citing_title":"FoNE: Precise Single-Token Number Embeddings via Fourier Features","ref_index":47,"is_internal_anchor":false},{"citing_arxiv_id":"2507.12549","citing_title":"The Serial Scaling Hypothesis","ref_index":137,"is_internal_anchor":false},{"citing_arxiv_id":"2602.01651","citing_title":"On the Spatiotemporal Dynamics of Generalization in Neural Networks","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2603.06870","citing_title":"LEAD: Breaking the No-Recovery Bottleneck in Long-Horizon Reasoning","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2603.29069","citing_title":"On the Mirage of Long-Range Dependency, with an Application to Integer Multiplication","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2604.25166","citing_title":"Training Transformers as a Universal Computer","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2403.07974","citing_title":"LiveCodeBench: Holistic and Contamination Free Evaluation of Large Language Models for Code","ref_index":94,"is_internal_anchor":false},{"citing_arxiv_id":"2604.15306","citing_title":"Generalization in LLM Problem Solving: The Case of the Shortest Path","ref_index":56,"is_internal_anchor":false},{"citing_arxiv_id":"2604.17857","citing_title":"On the Emergence of Syntax by Means of Local Interaction","ref_index":46,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/BRGD5OASOPAGUWAJAVEBFOQ52X","json":"https://pith.science/pith/BRGD5OASOPAGUWAJAVEBFOQ52X.json","graph_json":"https://pith.science/api/pith-number/BRGD5OASOPAGUWAJAVEBFOQ52X/graph.json","events_json":"https://pith.science/api/pith-number/BRGD5OASOPAGUWAJAVEBFOQ52X/events.json","paper":"https://pith.science/paper/BRGD5OAS"},"agent_actions":{"view_html":"https://pith.science/pith/BRGD5OASOPAGUWAJAVEBFOQ52X","download_json":"https://pith.science/pith/BRGD5OASOPAGUWAJAVEBFOQ52X.json","view_paper":"https://pith.science/paper/BRGD5OAS","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2310.16028&json=true","fetch_graph":"https://pith.science/api/pith-number/BRGD5OASOPAGUWAJAVEBFOQ52X/graph.json","fetch_events":"https://pith.science/api/pith-number/BRGD5OASOPAGUWAJAVEBFOQ52X/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/BRGD5OASOPAGUWAJAVEBFOQ52X/action/timestamp_anchor","attest_storage":"https://pith.science/pith/BRGD5OASOPAGUWAJAVEBFOQ52X/action/storage_attestation","attest_author":"https://pith.science/pith/BRGD5OASOPAGUWAJAVEBFOQ52X/action/author_attestation","sign_citation":"https://pith.science/pith/BRGD5OASOPAGUWAJAVEBFOQ52X/action/citation_signature","submit_replication":"https://pith.science/pith/BRGD5OASOPAGUWAJAVEBFOQ52X/action/replication_record"}},"created_at":"2026-07-05T07:04:33.998796+00:00","updated_at":"2026-07-05T07:04:33.998796+00:00"}