{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:S6NG5FS4S232RA5XXXAY227JPX","short_pith_number":"pith:S6NG5FS4","schema_version":"1.0","canonical_sha256":"979a6e965c96b7a883b7bdc18d6be97dc04c285bddd50a4a1d6ca449979ea7c4","source":{"kind":"arxiv","id":"2410.01405","version":7},"attestation_state":"computed","paper":{"title":"On Expressive Power of Looped Transformers: Theoretical Analysis and Enhancement via Timestep Encoding","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Issei Sato, Kevin Xu","submitted_at":"2024-10-02T10:31:17Z","abstract_excerpt":"Looped Transformers provide advantages in parameter efficiency, computational capabilities, and generalization for reasoning tasks. However, their expressive power regarding function approximation remains underexplored. In this paper, we establish the approximation rate of Looped Transformers by defining the modulus of continuity for sequence-to-sequence functions. This reveals a limitation specific to the looped architecture. That is, the analysis prompts the incorporation of scaling parameters for each loop, conditioned on timestep encoding. Experiments validate the theoretical results, show"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.01405","kind":"arxiv","version":7},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-10-02T10:31:17Z","cross_cats_sorted":[],"title_canon_sha256":"38d0f49e73431f486c367926218672b538e7053c83664829e1c51488f0f7e9ef","abstract_canon_sha256":"42197198b090cc1a1736d39c1a779220434cf8aa55ba8deb28091d330fdd717b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:16:29.656399Z","signature_b64":"xD08xEKYwREcWTvJhyvOHRI7VJ7Z9mbUwLIx+9jImSe1JOG5EM8RixH0ddCLaziAP1HDgA1qg5Ju890An5UKAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"979a6e965c96b7a883b7bdc18d6be97dc04c285bddd50a4a1d6ca449979ea7c4","last_reissued_at":"2026-07-05T11:16:29.655883Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:16:29.655883Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"On Expressive Power of Looped Transformers: Theoretical Analysis and Enhancement via Timestep Encoding","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Issei Sato, Kevin Xu","submitted_at":"2024-10-02T10:31:17Z","abstract_excerpt":"Looped Transformers provide advantages in parameter efficiency, computational capabilities, and generalization for reasoning tasks. However, their expressive power regarding function approximation remains underexplored. In this paper, we establish the approximation rate of Looped Transformers by defining the modulus of continuity for sequence-to-sequence functions. This reveals a limitation specific to the looped architecture. That is, the analysis prompts the incorporation of scaling parameters for each loop, conditioned on timestep encoding. Experiments validate the theoretical results, show"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.01405","kind":"arxiv","version":7},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.01405/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.01405","created_at":"2026-07-05T11:16:29.655948+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.01405v7","created_at":"2026-07-05T11:16:29.655948+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.01405","created_at":"2026-07-05T11:16:29.655948+00:00"},{"alias_kind":"pith_short_12","alias_value":"S6NG5FS4S232","created_at":"2026-07-05T11:16:29.655948+00:00"},{"alias_kind":"pith_short_16","alias_value":"S6NG5FS4S232RA5X","created_at":"2026-07-05T11:16:29.655948+00:00"},{"alias_kind":"pith_short_8","alias_value":"S6NG5FS4","created_at":"2026-07-05T11:16:29.655948+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.30229","citing_title":"Anti Mode-Collapse in Mean-Field Transformer via Auxiliary Variables","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18797","citing_title":"Simply Stabilizing the Loop via Fully Looped Transformer","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2510.03206","citing_title":"Coevolutionary Continuous Discrete Diffusion: Make Your Diffusion Language Model a Latent Reasoner","ref_index":40,"is_internal_anchor":false},{"citing_arxiv_id":"2604.19550","citing_title":"LoopCTR: Unlocking the Loop Scaling Power for Click-Through Rate Prediction","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2604.11791","citing_title":"A Mechanistic Analysis of Looped Reasoning Language Models","ref_index":31,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/S6NG5FS4S232RA5XXXAY227JPX","json":"https://pith.science/pith/S6NG5FS4S232RA5XXXAY227JPX.json","graph_json":"https://pith.science/api/pith-number/S6NG5FS4S232RA5XXXAY227JPX/graph.json","events_json":"https://pith.science/api/pith-number/S6NG5FS4S232RA5XXXAY227JPX/events.json","paper":"https://pith.science/paper/S6NG5FS4"},"agent_actions":{"view_html":"https://pith.science/pith/S6NG5FS4S232RA5XXXAY227JPX","download_json":"https://pith.science/pith/S6NG5FS4S232RA5XXXAY227JPX.json","view_paper":"https://pith.science/paper/S6NG5FS4","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.01405&json=true","fetch_graph":"https://pith.science/api/pith-number/S6NG5FS4S232RA5XXXAY227JPX/graph.json","fetch_events":"https://pith.science/api/pith-number/S6NG5FS4S232RA5XXXAY227JPX/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/S6NG5FS4S232RA5XXXAY227JPX/action/timestamp_anchor","attest_storage":"https://pith.science/pith/S6NG5FS4S232RA5XXXAY227JPX/action/storage_attestation","attest_author":"https://pith.science/pith/S6NG5FS4S232RA5XXXAY227JPX/action/author_attestation","sign_citation":"https://pith.science/pith/S6NG5FS4S232RA5XXXAY227JPX/action/citation_signature","submit_replication":"https://pith.science/pith/S6NG5FS4S232RA5XXXAY227JPX/action/replication_record"}},"created_at":"2026-07-05T11:16:29.655948+00:00","updated_at":"2026-07-05T11:16:29.655948+00:00"}