{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:JQ3XOKAU7BTZB4BO2P5RTH6WH6","short_pith_number":"pith:JQ3XOKAU","schema_version":"1.0","canonical_sha256":"4c37772814f86790f02ed3fb199fd63fb987d77f5f045c9d20b3a97e7f11347f","source":{"kind":"arxiv","id":"2504.12175","version":1},"attestation_state":"computed","paper":{"title":"Approximation Bounds for Transformer Networks with Application to Regression","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"stat.ML","authors_text":"Bokai Yan, Defeng Sun, Yang Wang, Yanming Lai, Yuling Jiao","submitted_at":"2025-04-16T15:25:58Z","abstract_excerpt":"We explore the approximation capabilities of Transformer networks for H\\\"older and Sobolev functions, and apply these results to address nonparametric regression estimation with dependent observations. First, we establish novel upper bounds for standard Transformer networks approximating sequence-to-sequence mappings whose component functions are H\\\"older continuous with smoothness index $\\gamma \\in (0,1]$. To achieve an approximation error $\\varepsilon$ under the $L^p$-norm for $p \\in [1, \\infty]$, it suffices to use a fixed-depth Transformer network whose total number of parameters scales as"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2504.12175","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"stat.ML","submitted_at":"2025-04-16T15:25:58Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"2342e201a7ea55929da03bbe59b2c69c3aaa62b1263654ee58279caa9eeefa42","abstract_canon_sha256":"c3447c700e73c535031153e6c2fac91665c6c5044088ff66bbb5b47b09c078fc"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:50:03.826359Z","signature_b64":"y1O325aaJhji81chyF9aDMHMjqouxQnvSx0LvbZvdonnXfj2g0Hpr1hJWCU3H6col/tibNuqjhLYdfmAnceUDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"4c37772814f86790f02ed3fb199fd63fb987d77f5f045c9d20b3a97e7f11347f","last_reissued_at":"2026-07-05T10:50:03.825838Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:50:03.825838Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Approximation Bounds for Transformer Networks with Application to Regression","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"stat.ML","authors_text":"Bokai Yan, Defeng Sun, Yang Wang, Yanming Lai, Yuling Jiao","submitted_at":"2025-04-16T15:25:58Z","abstract_excerpt":"We explore the approximation capabilities of Transformer networks for H\\\"older and Sobolev functions, and apply these results to address nonparametric regression estimation with dependent observations. First, we establish novel upper bounds for standard Transformer networks approximating sequence-to-sequence mappings whose component functions are H\\\"older continuous with smoothness index $\\gamma \\in (0,1]$. To achieve an approximation error $\\varepsilon$ under the $L^p$-norm for $p \\in [1, \\infty]$, it suffices to use a fixed-depth Transformer network whose total number of parameters scales as"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2504.12175","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2504.12175/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2504.12175","created_at":"2026-07-05T10:50:03.825893+00:00"},{"alias_kind":"arxiv_version","alias_value":"2504.12175v1","created_at":"2026-07-05T10:50:03.825893+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2504.12175","created_at":"2026-07-05T10:50:03.825893+00:00"},{"alias_kind":"pith_short_12","alias_value":"JQ3XOKAU7BTZ","created_at":"2026-07-05T10:50:03.825893+00:00"},{"alias_kind":"pith_short_16","alias_value":"JQ3XOKAU7BTZB4BO","created_at":"2026-07-05T10:50:03.825893+00:00"},{"alias_kind":"pith_short_8","alias_value":"JQ3XOKAU","created_at":"2026-07-05T10:50:03.825893+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.06781","citing_title":"On Explicit Super-Expressive Approximation for Neural Networks","ref_index":124,"is_internal_anchor":true},{"citing_arxiv_id":"2605.07463","citing_title":"Approximation Error Upper and Lower Bounds for H\\\"{o}lder Class with Transformers","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07463","citing_title":"Approximation Error Upper and Lower Bounds for H\\\"{o}lder Class with Transformers","ref_index":17,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/JQ3XOKAU7BTZB4BO2P5RTH6WH6","json":"https://pith.science/pith/JQ3XOKAU7BTZB4BO2P5RTH6WH6.json","graph_json":"https://pith.science/api/pith-number/JQ3XOKAU7BTZB4BO2P5RTH6WH6/graph.json","events_json":"https://pith.science/api/pith-number/JQ3XOKAU7BTZB4BO2P5RTH6WH6/events.json","paper":"https://pith.science/paper/JQ3XOKAU"},"agent_actions":{"view_html":"https://pith.science/pith/JQ3XOKAU7BTZB4BO2P5RTH6WH6","download_json":"https://pith.science/pith/JQ3XOKAU7BTZB4BO2P5RTH6WH6.json","view_paper":"https://pith.science/paper/JQ3XOKAU","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2504.12175&json=true","fetch_graph":"https://pith.science/api/pith-number/JQ3XOKAU7BTZB4BO2P5RTH6WH6/graph.json","fetch_events":"https://pith.science/api/pith-number/JQ3XOKAU7BTZB4BO2P5RTH6WH6/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/JQ3XOKAU7BTZB4BO2P5RTH6WH6/action/timestamp_anchor","attest_storage":"https://pith.science/pith/JQ3XOKAU7BTZB4BO2P5RTH6WH6/action/storage_attestation","attest_author":"https://pith.science/pith/JQ3XOKAU7BTZB4BO2P5RTH6WH6/action/author_attestation","sign_citation":"https://pith.science/pith/JQ3XOKAU7BTZB4BO2P5RTH6WH6/action/citation_signature","submit_replication":"https://pith.science/pith/JQ3XOKAU7BTZB4BO2P5RTH6WH6/action/replication_record"}},"created_at":"2026-07-05T10:50:03.825893+00:00","updated_at":"2026-07-05T10:50:03.825893+00:00"}