{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2019:ITFBBX2Y4QTGWPUTJKGRI2AXTF","short_pith_number":"pith:ITFBBX2Y","schema_version":"1.0","canonical_sha256":"44ca10df58e4266b3e934a8d146817994eb89f77ecf5a7d6c068ffd8258a67c0","source":{"kind":"arxiv","id":"1912.10077","version":2},"attestation_state":"computed","paper":{"title":"Are Transformers universal approximators of sequence-to-sequence functions?","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["stat.ML"],"primary_cat":"cs.LG","authors_text":"Ankit Singh Rawat, Chulhee Yun, Sanjiv Kumar, Sashank J. Reddi, Srinadh Bhojanapalli","submitted_at":"2019-12-20T19:49:32Z","abstract_excerpt":"Despite the widespread adoption of Transformer models for NLP tasks, the expressive power of these models is not well-understood. In this paper, we establish that Transformer models are universal approximators of continuous permutation equivariant sequence-to-sequence functions with compact support, which is quite surprising given the amount of shared parameters in these models. Furthermore, using positional encodings, we circumvent the restriction of permutation equivariance, and show that Transformer models can universally approximate arbitrary continuous sequence-to-sequence functions on a "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"1912.10077","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2019-12-20T19:49:32Z","cross_cats_sorted":["stat.ML"],"title_canon_sha256":"d7cb441cb479e7a1d4ee538f376ac5a57c869bab985d68cfd418962c549ae22d","abstract_canon_sha256":"fd5747e382e698b86361fdb01a96ed36de82f0f055f4e70dec9b8ad6d1eb1cad"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T00:43:21.543694Z","signature_b64":"BPe1VY2OADlrAneYJdD+1Tuob8DxwsEGkr9JOVR/6jXXs84JK3BGEydrjMl3iEjMyHvE3uZR3QRn7Jzo4AKgDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"44ca10df58e4266b3e934a8d146817994eb89f77ecf5a7d6c068ffd8258a67c0","last_reissued_at":"2026-07-05T00:43:21.543204Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T00:43:21.543204Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Are Transformers universal approximators of sequence-to-sequence functions?","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["stat.ML"],"primary_cat":"cs.LG","authors_text":"Ankit Singh Rawat, Chulhee Yun, Sanjiv Kumar, Sashank J. Reddi, Srinadh Bhojanapalli","submitted_at":"2019-12-20T19:49:32Z","abstract_excerpt":"Despite the widespread adoption of Transformer models for NLP tasks, the expressive power of these models is not well-understood. In this paper, we establish that Transformer models are universal approximators of continuous permutation equivariant sequence-to-sequence functions with compact support, which is quite surprising given the amount of shared parameters in these models. Furthermore, using positional encodings, we circumvent the restriction of permutation equivariance, and show that Transformer models can universally approximate arbitrary continuous sequence-to-sequence functions on a "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1912.10077","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/1912.10077/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"1912.10077","created_at":"2026-07-05T00:43:21.543263+00:00"},{"alias_kind":"arxiv_version","alias_value":"1912.10077v2","created_at":"2026-07-05T00:43:21.543263+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1912.10077","created_at":"2026-07-05T00:43:21.543263+00:00"},{"alias_kind":"pith_short_12","alias_value":"ITFBBX2Y4QTG","created_at":"2026-07-05T00:43:21.543263+00:00"},{"alias_kind":"pith_short_16","alias_value":"ITFBBX2Y4QTGWPUT","created_at":"2026-07-05T00:43:21.543263+00:00"},{"alias_kind":"pith_short_8","alias_value":"ITFBBX2Y","created_at":"2026-07-05T00:43:21.543263+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":10,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.23364","citing_title":"Convergence of Gradient Descent for General Neural Network Architectures Beyond the NTK Regime","ref_index":110,"is_internal_anchor":false},{"citing_arxiv_id":"2606.05030","citing_title":"Imbuing Large Language Models with Bidirectional Logic for Robust Chain Repair","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2605.24124","citing_title":"A generative pre-trained transformer with Kerr-soliton attention","ref_index":37,"is_internal_anchor":false},{"citing_arxiv_id":"2606.04010","citing_title":"The Variance Brain Foundation Models Forgot: Third-Order Statistics Predict Cognition Where Billion-Parameter Models Fail","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2508.16745","citing_title":"Beyond Memorization: Extending Reasoning Depth with Recurrence, Memory and Test-Time Compute Scaling","ref_index":76,"is_internal_anchor":false},{"citing_arxiv_id":"2510.23912","citing_title":"Key and Value Weights Are Probably All You Need: On the Necessity of the Query, Key, Value weight Triplet in Self-Attention Transformers","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2604.24154","citing_title":"Progressive Approximation in Deep Residual Networks: Theory and Validation","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06826","citing_title":"How Does Attention Help? Insights from Random Matrices on Signal Recovery from Sequence Models","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2604.14702","citing_title":"Gating Enables Curvature: A Geometric Expressivity Gap in Attention","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2604.16653","citing_title":"Continuous transformations of probability measures and their transport representations","ref_index":27,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ITFBBX2Y4QTGWPUTJKGRI2AXTF","json":"https://pith.science/pith/ITFBBX2Y4QTGWPUTJKGRI2AXTF.json","graph_json":"https://pith.science/api/pith-number/ITFBBX2Y4QTGWPUTJKGRI2AXTF/graph.json","events_json":"https://pith.science/api/pith-number/ITFBBX2Y4QTGWPUTJKGRI2AXTF/events.json","paper":"https://pith.science/paper/ITFBBX2Y"},"agent_actions":{"view_html":"https://pith.science/pith/ITFBBX2Y4QTGWPUTJKGRI2AXTF","download_json":"https://pith.science/pith/ITFBBX2Y4QTGWPUTJKGRI2AXTF.json","view_paper":"https://pith.science/paper/ITFBBX2Y","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=1912.10077&json=true","fetch_graph":"https://pith.science/api/pith-number/ITFBBX2Y4QTGWPUTJKGRI2AXTF/graph.json","fetch_events":"https://pith.science/api/pith-number/ITFBBX2Y4QTGWPUTJKGRI2AXTF/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ITFBBX2Y4QTGWPUTJKGRI2AXTF/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ITFBBX2Y4QTGWPUTJKGRI2AXTF/action/storage_attestation","attest_author":"https://pith.science/pith/ITFBBX2Y4QTGWPUTJKGRI2AXTF/action/author_attestation","sign_citation":"https://pith.science/pith/ITFBBX2Y4QTGWPUTJKGRI2AXTF/action/citation_signature","submit_replication":"https://pith.science/pith/ITFBBX2Y4QTGWPUTJKGRI2AXTF/action/replication_record"}},"created_at":"2026-07-05T00:43:21.543263+00:00","updated_at":"2026-07-05T00:43:21.543263+00:00"}