{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:3KQZCGFOMEM4UVYVF2KXQR7ZA2","short_pith_number":"pith:3KQZCGFO","schema_version":"1.0","canonical_sha256":"daa19118ae6119ca57152e957847f906b4403613e3b7c5a7836242c0f5be6d1d","source":{"kind":"arxiv","id":"2409.15647","version":5},"attestation_state":"computed","paper":{"title":"Looped Transformers for Length Generalization","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Kangwook Lee, Kannan Ramchandran, Yilun Du, Ying Fan","submitted_at":"2024-09-24T01:21:17Z","abstract_excerpt":"Recent work has shown that Transformers trained from scratch can successfully solve various arithmetic and algorithmic tasks, such as adding numbers and computing parity. While these Transformers generalize well on unseen inputs of the same length, they struggle with length generalization, i.e., handling inputs of unseen lengths. In this work, we demonstrate that looped Transformers with an adaptive number of steps significantly improve length generalization. We focus on tasks with a known iterative solution, involving multiple iterations of a RASP-L operation - a length-generalizable operatio"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2409.15647","kind":"arxiv","version":5},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-09-24T01:21:17Z","cross_cats_sorted":[],"title_canon_sha256":"7a565edae318bc5c5149079ff3ca46bf11cc199fa8848734b7cee5e45de380ed","abstract_canon_sha256":"f31dffe6149cfba08c497d2add8fbe512414d3f20421c1054a91429d1be03aff"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:01:27.003327Z","signature_b64":"GZiBMCowEmT5u/SlYTzD+T7yUlnoLzu5HP5N49mvGEipQ/XOFqS8Ww0yNvBx57d4oU+nqD4moTHcSVWMX+Z6AA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"daa19118ae6119ca57152e957847f906b4403613e3b7c5a7836242c0f5be6d1d","last_reissued_at":"2026-07-05T11:01:27.002748Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:01:27.002748Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Looped Transformers for Length Generalization","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Kangwook Lee, Kannan Ramchandran, Yilun Du, Ying Fan","submitted_at":"2024-09-24T01:21:17Z","abstract_excerpt":"Recent work has shown that Transformers trained from scratch can successfully solve various arithmetic and algorithmic tasks, such as adding numbers and computing parity. While these Transformers generalize well on unseen inputs of the same length, they struggle with length generalization, i.e., handling inputs of unseen lengths. In this work, we demonstrate that looped Transformers with an adaptive number of steps significantly improve length generalization. We focus on tasks with a known iterative solution, involving multiple iterations of a RASP-L operation - a length-generalizable operatio"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2409.15647","kind":"arxiv","version":5},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2409.15647/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2409.15647","created_at":"2026-07-05T11:01:27.002806+00:00"},{"alias_kind":"arxiv_version","alias_value":"2409.15647v5","created_at":"2026-07-05T11:01:27.002806+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2409.15647","created_at":"2026-07-05T11:01:27.002806+00:00"},{"alias_kind":"pith_short_12","alias_value":"3KQZCGFOMEM4","created_at":"2026-07-05T11:01:27.002806+00:00"},{"alias_kind":"pith_short_16","alias_value":"3KQZCGFOMEM4UVYV","created_at":"2026-07-05T11:01:27.002806+00:00"},{"alias_kind":"pith_short_8","alias_value":"3KQZCGFO","created_at":"2026-07-05T11:01:27.002806+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":18,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.25008","citing_title":"Neural Scaling Universality: If Exponents Are Fixed, Time to Understand Coefficients","ref_index":46,"is_internal_anchor":false},{"citing_arxiv_id":"2606.20737","citing_title":"Repeated Shared Access Enables Grokking, but Edit Propagation Depends on an Addressable Memory","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2606.18022","citing_title":"Recursive Scaling in Masked Diffusion Models","ref_index":47,"is_internal_anchor":false},{"citing_arxiv_id":"2606.06574","citing_title":"Skip a Layer or Loop It? Learning Program-of-Layers in LLMs","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2605.26733","citing_title":"Stabilizing Recurrent Dynamics for Test-Time Scalable Latent Reasoning in Looped Language Models","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2605.26106","citing_title":"Looped Diffusion Language Models","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2605.30229","citing_title":"Anti Mode-Collapse in Mean-Field Transformer via Auxiliary Variables","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18797","citing_title":"Simply Stabilizing the Loop via Fully Looped Transformer","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2605.16343","citing_title":"LoopQ: Quantization for Recursive Transformers","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2510.03206","citing_title":"Coevolutionary Continuous Discrete Diffusion: Make Your Diffusion Language Model a Latent Reasoner","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2603.29057","citing_title":"LA-Sign: Looped Transformers with Geometry-aware Alignment for Skeleton-based Sign Language Recognition","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2603.29069","citing_title":"On the Mirage of Long-Range Dependency, with an Application to Integer Multiplication","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2604.01577","citing_title":"Exploration of Fast-Slow Latent Recurrence for Train-Short, Test-Long Generalization","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2604.19550","citing_title":"LoopCTR: Unlocking the Loop Scaling Power for Click-Through Rate Prediction","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2412.06769","citing_title":"Training Large Language Models to Reason in a Continuous Latent Space","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2604.12426","citing_title":"Do Transformers Use their Depth Adaptively? Evidence from a Relational Reasoning Task","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2604.09168","citing_title":"ELT: Elastic Looped Transformers for Visual Generation","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2604.15306","citing_title":"Generalization in LLM Problem Solving: The Case of the Shortest Path","ref_index":17,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/3KQZCGFOMEM4UVYVF2KXQR7ZA2","json":"https://pith.science/pith/3KQZCGFOMEM4UVYVF2KXQR7ZA2.json","graph_json":"https://pith.science/api/pith-number/3KQZCGFOMEM4UVYVF2KXQR7ZA2/graph.json","events_json":"https://pith.science/api/pith-number/3KQZCGFOMEM4UVYVF2KXQR7ZA2/events.json","paper":"https://pith.science/paper/3KQZCGFO"},"agent_actions":{"view_html":"https://pith.science/pith/3KQZCGFOMEM4UVYVF2KXQR7ZA2","download_json":"https://pith.science/pith/3KQZCGFOMEM4UVYVF2KXQR7ZA2.json","view_paper":"https://pith.science/paper/3KQZCGFO","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2409.15647&json=true","fetch_graph":"https://pith.science/api/pith-number/3KQZCGFOMEM4UVYVF2KXQR7ZA2/graph.json","fetch_events":"https://pith.science/api/pith-number/3KQZCGFOMEM4UVYVF2KXQR7ZA2/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/3KQZCGFOMEM4UVYVF2KXQR7ZA2/action/timestamp_anchor","attest_storage":"https://pith.science/pith/3KQZCGFOMEM4UVYVF2KXQR7ZA2/action/storage_attestation","attest_author":"https://pith.science/pith/3KQZCGFOMEM4UVYVF2KXQR7ZA2/action/author_attestation","sign_citation":"https://pith.science/pith/3KQZCGFOMEM4UVYVF2KXQR7ZA2/action/citation_signature","submit_replication":"https://pith.science/pith/3KQZCGFOMEM4UVYVF2KXQR7ZA2/action/replication_record"}},"created_at":"2026-07-05T11:01:27.002806+00:00","updated_at":"2026-07-05T11:01:27.002806+00:00"}