{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:AREAWPEESKNG76B357YNC5WSI2","short_pith_number":"pith:AREAWPEE","schema_version":"1.0","canonical_sha256":"04480b3c84929a6ff83beff0d176d246b66d81641ee52dc1e11a7e0ea46714a9","source":{"kind":"arxiv","id":"2303.06053","version":5},"attestation_state":"computed","paper":{"title":"TSMixer: An All-MLP Architecture for Time Series Forecasting","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Chun-Liang Li, Nate Yoder, Sercan O. Arik, Si-An Chen, Tomas Pfister","submitted_at":"2023-03-10T16:41:24Z","abstract_excerpt":"Real-world time-series datasets are often multivariate with complex dynamics. To capture this complexity, high capacity architectures like recurrent- or attention-based sequential deep learning models have become popular. However, recent work demonstrates that simple univariate linear models can outperform such deep learning models on several commonly used academic benchmarks. Extending them, in this paper, we investigate the capabilities of linear models for time-series forecasting and present Time-Series Mixer (TSMixer), a novel architecture designed by stacking multi-layer perceptrons (MLPs"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2303.06053","kind":"arxiv","version":5},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2023-03-10T16:41:24Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"8e489f3e416b71007df8a539c70239a2f80e48221532ba0d137fc5de8e39b048","abstract_canon_sha256":"3388243c6f47ce06e434915f9d0dbd2feed32efec5dee7c942a010a1f6246d10"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:49:19.983796Z","signature_b64":"0xBYszyR6L/CWide450JUFnTDD2PqJ6g1c92PppNffdRSGOMgS0anLs7Th8qLVGAxDtdq9vo7kZAPHMwE2fzBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"04480b3c84929a6ff83beff0d176d246b66d81641ee52dc1e11a7e0ea46714a9","last_reissued_at":"2026-07-05T06:49:19.983290Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:49:19.983290Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"TSMixer: An All-MLP Architecture for Time Series Forecasting","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Chun-Liang Li, Nate Yoder, Sercan O. Arik, Si-An Chen, Tomas Pfister","submitted_at":"2023-03-10T16:41:24Z","abstract_excerpt":"Real-world time-series datasets are often multivariate with complex dynamics. To capture this complexity, high capacity architectures like recurrent- or attention-based sequential deep learning models have become popular. However, recent work demonstrates that simple univariate linear models can outperform such deep learning models on several commonly used academic benchmarks. Extending them, in this paper, we investigate the capabilities of linear models for time-series forecasting and present Time-Series Mixer (TSMixer), a novel architecture designed by stacking multi-layer perceptrons (MLPs"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2303.06053","kind":"arxiv","version":5},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2303.06053/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2303.06053","created_at":"2026-07-05T06:49:19.983358+00:00"},{"alias_kind":"arxiv_version","alias_value":"2303.06053v5","created_at":"2026-07-05T06:49:19.983358+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2303.06053","created_at":"2026-07-05T06:49:19.983358+00:00"},{"alias_kind":"pith_short_12","alias_value":"AREAWPEESKNG","created_at":"2026-07-05T06:49:19.983358+00:00"},{"alias_kind":"pith_short_16","alias_value":"AREAWPEESKNG76B3","created_at":"2026-07-05T06:49:19.983358+00:00"},{"alias_kind":"pith_short_8","alias_value":"AREAWPEE","created_at":"2026-07-05T06:49:19.983358+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":15,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.27282","citing_title":"How Good Can Linear Models Be for Time-Series Forecasting?","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2606.20010","citing_title":"Self-Adaptive Scale Handling for Forecasting Time Series with Scale Heterogeneity","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2606.19560","citing_title":"Understanding Key Features of Time Series Foundation Models from Epidemic Forecasting","ref_index":55,"is_internal_anchor":false},{"citing_arxiv_id":"2606.31470","citing_title":"CLOUDADV: Decision-Aligned Instance Sizing with Zero-Shot Foundation Models under Drift","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2606.27282","citing_title":"How Good Can Linear Models Be for Time-Series Forecasting?","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2605.28166","citing_title":"QuITE: Query-Based Irregular Time Series Embedding","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2606.00338","citing_title":"CHAM-net: A Contrastive Hierarchical Adaptive Meta-network for Robust Global Methane Flux Prediction","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2305.10721","citing_title":"Revisiting Long-term Time Series Forecasting: An Investigation on Linear Mapping","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2407.13278","citing_title":"Deep Time Series Models: A Comprehensive Survey and Benchmark","ref_index":96,"is_internal_anchor":false},{"citing_arxiv_id":"2505.16786","citing_title":"FlowMixer: A Depth-Agnostic Neural Architecture for Interpretable Spatiotemporal Forecasting","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2605.22086","citing_title":"GenHAR: Generalizing Cross-domain Human Activity Recognition for Last-mile Delivery","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2509.14000","citing_title":"JaGuard: Position Error Correction of GNSS Jamming with Deep Temporal Graphs","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2310.10688","citing_title":"A decoder-only foundation model for time-series forecasting","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2605.13678","citing_title":"Three-Stage Learning Unlocks Strong Performance in Simple Models for Long-Term Time Series Forecasting","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2604.26762","citing_title":"Exploring the Potential of Probabilistic Transformer for Time Series Modeling: A Report on the ST-PT Framework","ref_index":29,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/AREAWPEESKNG76B357YNC5WSI2","json":"https://pith.science/pith/AREAWPEESKNG76B357YNC5WSI2.json","graph_json":"https://pith.science/api/pith-number/AREAWPEESKNG76B357YNC5WSI2/graph.json","events_json":"https://pith.science/api/pith-number/AREAWPEESKNG76B357YNC5WSI2/events.json","paper":"https://pith.science/paper/AREAWPEE"},"agent_actions":{"view_html":"https://pith.science/pith/AREAWPEESKNG76B357YNC5WSI2","download_json":"https://pith.science/pith/AREAWPEESKNG76B357YNC5WSI2.json","view_paper":"https://pith.science/paper/AREAWPEE","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2303.06053&json=true","fetch_graph":"https://pith.science/api/pith-number/AREAWPEESKNG76B357YNC5WSI2/graph.json","fetch_events":"https://pith.science/api/pith-number/AREAWPEESKNG76B357YNC5WSI2/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/AREAWPEESKNG76B357YNC5WSI2/action/timestamp_anchor","attest_storage":"https://pith.science/pith/AREAWPEESKNG76B357YNC5WSI2/action/storage_attestation","attest_author":"https://pith.science/pith/AREAWPEESKNG76B357YNC5WSI2/action/author_attestation","sign_citation":"https://pith.science/pith/AREAWPEESKNG76B357YNC5WSI2/action/citation_signature","submit_replication":"https://pith.science/pith/AREAWPEESKNG76B357YNC5WSI2/action/replication_record"}},"created_at":"2026-07-05T06:49:19.983358+00:00","updated_at":"2026-07-05T06:49:19.983358+00:00"}