{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:P4NY7OLZSWPKRUT4SZCOYK7L4Z","short_pith_number":"pith:P4NY7OLZ","schema_version":"1.0","canonical_sha256":"7f1b8fb979959ea8d27c9644ec2bebe6771630e060410a41a643c1fbc699bc07","source":{"kind":"arxiv","id":"2311.12424","version":3},"attestation_state":"computed","paper":{"title":"Looped Transformers are Better at Learning Learning Algorithms","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.NE"],"primary_cat":"cs.LG","authors_text":"Dimitris Papailiopoulos, Kangwook Lee, Liu Yang, Robert Nowak","submitted_at":"2023-11-21T08:32:38Z","abstract_excerpt":"Transformers have demonstrated effectiveness in in-context solving data-fitting problems from various (latent) models, as reported by Garg et al. However, the absence of an inherent iterative structure in the transformer architecture presents a challenge in emulating the iterative algorithms, which are commonly employed in traditional machine learning methods. To address this, we propose the utilization of looped transformer architecture and its associated training methodology, with the aim of incorporating iterative characteristics into the transformer architectures. Experimental results sugg"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2311.12424","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2023-11-21T08:32:38Z","cross_cats_sorted":["cs.NE"],"title_canon_sha256":"648d014fab3ba52d17d062fbf469110c92d96f31431c58743d7ba9249722c2cb","abstract_canon_sha256":"f86a8fb72cc58386bdf4e3d73f3a242ff32de105be22e728f416954b75d76884"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:56:56.705436Z","signature_b64":"TjcizOFX+iyEI18NlBzpciNgUgNZU7WyINLNeiJt6PFP/GRjzzsiZH69cruellbrG/z7n8ONCiv4updhdIzhCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"7f1b8fb979959ea8d27c9644ec2bebe6771630e060410a41a643c1fbc699bc07","last_reissued_at":"2026-07-05T07:56:56.705004Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:56:56.705004Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Looped Transformers are Better at Learning Learning Algorithms","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.NE"],"primary_cat":"cs.LG","authors_text":"Dimitris Papailiopoulos, Kangwook Lee, Liu Yang, Robert Nowak","submitted_at":"2023-11-21T08:32:38Z","abstract_excerpt":"Transformers have demonstrated effectiveness in in-context solving data-fitting problems from various (latent) models, as reported by Garg et al. However, the absence of an inherent iterative structure in the transformer architecture presents a challenge in emulating the iterative algorithms, which are commonly employed in traditional machine learning methods. To address this, we propose the utilization of looped transformer architecture and its associated training methodology, with the aim of incorporating iterative characteristics into the transformer architectures. Experimental results sugg"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2311.12424","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2311.12424/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2311.12424","created_at":"2026-07-05T07:56:56.705063+00:00"},{"alias_kind":"arxiv_version","alias_value":"2311.12424v3","created_at":"2026-07-05T07:56:56.705063+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2311.12424","created_at":"2026-07-05T07:56:56.705063+00:00"},{"alias_kind":"pith_short_12","alias_value":"P4NY7OLZSWPK","created_at":"2026-07-05T07:56:56.705063+00:00"},{"alias_kind":"pith_short_16","alias_value":"P4NY7OLZSWPKRUT4","created_at":"2026-07-05T07:56:56.705063+00:00"},{"alias_kind":"pith_short_8","alias_value":"P4NY7OLZ","created_at":"2026-07-05T07:56:56.705063+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":23,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.18022","citing_title":"Recursive Scaling in Masked Diffusion Models","ref_index":48,"is_internal_anchor":false},{"citing_arxiv_id":"2606.18208","citing_title":"Looped World Models","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2606.06574","citing_title":"Skip a Layer or Loop It? Learning Program-of-Layers in LLMs","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2606.06624","citing_title":"Principles and Practice of Deep Representation Learning: or a Mathematical Theory of Memory","ref_index":114,"is_internal_anchor":false},{"citing_arxiv_id":"2605.26733","citing_title":"Stabilizing Recurrent Dynamics for Test-Time Scalable Latent Reasoning in Looped Language Models","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2606.29983","citing_title":"Stabilizing Extrapolation in Looped Transformers via Learned Stochastic Stopping","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2605.26106","citing_title":"Looped Diffusion Language Models","ref_index":74,"is_internal_anchor":false},{"citing_arxiv_id":"2606.18208","citing_title":"Looped World Models","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2510.12773","citing_title":"Dr.LLM: Dynamic Layer Routing in LLMs","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2605.19376","citing_title":"Generative Recursive Reasoning","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18797","citing_title":"Simply Stabilizing the Loop via Fully Looped Transformer","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07721","citing_title":"Memory-Efficient Looped Transformer: Decoupling Compute from Memory in Looped Language Models","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2605.19376","citing_title":"Generative Recursive Reasoning","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2508.16745","citing_title":"Beyond Memorization: Extending Reasoning Depth with Recurrence, Memory and Test-Time Compute Scaling","ref_index":73,"is_internal_anchor":false},{"citing_arxiv_id":"2510.25741","citing_title":"Scaling Latent Reasoning via Looped Language Models","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2604.21254","citing_title":"Hyperloop Transformers","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2604.12946","citing_title":"Parcae: Scaling Laws For Stable Looped Language Models","ref_index":88,"is_internal_anchor":false},{"citing_arxiv_id":"2604.11791","citing_title":"A Mechanistic Analysis of Looped Reasoning Language Models","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2604.09168","citing_title":"ELT: Elastic Looped Transformers for Visual Generation","ref_index":79,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07588","citing_title":"Revisiting Transformer Layer Parameterization Through Causal Energy Minimization","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07721","citing_title":"Memory-Efficient Looped Transformer: Decoupling Compute from Memory in Looped Language Models","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2604.15259","citing_title":"Stability and Generalization in Looped Transformers","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2604.18839","citing_title":"One Step Forward and K Steps Back: Better Reasoning with Denoising Recursion Models","ref_index":213,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/P4NY7OLZSWPKRUT4SZCOYK7L4Z","json":"https://pith.science/pith/P4NY7OLZSWPKRUT4SZCOYK7L4Z.json","graph_json":"https://pith.science/api/pith-number/P4NY7OLZSWPKRUT4SZCOYK7L4Z/graph.json","events_json":"https://pith.science/api/pith-number/P4NY7OLZSWPKRUT4SZCOYK7L4Z/events.json","paper":"https://pith.science/paper/P4NY7OLZ"},"agent_actions":{"view_html":"https://pith.science/pith/P4NY7OLZSWPKRUT4SZCOYK7L4Z","download_json":"https://pith.science/pith/P4NY7OLZSWPKRUT4SZCOYK7L4Z.json","view_paper":"https://pith.science/paper/P4NY7OLZ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2311.12424&json=true","fetch_graph":"https://pith.science/api/pith-number/P4NY7OLZSWPKRUT4SZCOYK7L4Z/graph.json","fetch_events":"https://pith.science/api/pith-number/P4NY7OLZSWPKRUT4SZCOYK7L4Z/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/P4NY7OLZSWPKRUT4SZCOYK7L4Z/action/timestamp_anchor","attest_storage":"https://pith.science/pith/P4NY7OLZSWPKRUT4SZCOYK7L4Z/action/storage_attestation","attest_author":"https://pith.science/pith/P4NY7OLZSWPKRUT4SZCOYK7L4Z/action/author_attestation","sign_citation":"https://pith.science/pith/P4NY7OLZSWPKRUT4SZCOYK7L4Z/action/citation_signature","submit_replication":"https://pith.science/pith/P4NY7OLZSWPKRUT4SZCOYK7L4Z/action/replication_record"}},"created_at":"2026-07-05T07:56:56.705063+00:00","updated_at":"2026-07-05T07:56:56.705063+00:00"}