{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:OVJ6ZYCYZO2UDQL57G4T7YWRM2","short_pith_number":"pith:OVJ6ZYCY","schema_version":"1.0","canonical_sha256":"7553ece058cbb541c17df9b93fe2d166bbb5ff74810f3c7fe073f5844f5a2ff7","source":{"kind":"arxiv","id":"2402.10198","version":3},"attestation_state":"computed","paper":{"title":"SAMformer: Unlocking the Potential of Transformers in Time Series Forecasting with Sharpness-Aware Minimization and Channel-Wise Attention","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["stat.ML"],"primary_cat":"cs.LG","authors_text":"Aladin Virmaux, Ambroise Odonnat, Giuseppe Paolo, Ievgen Redko, Romain Ilbert, Themis Palpanas, Vasilii Feofanov","submitted_at":"2024-02-15T18:55:05Z","abstract_excerpt":"Transformer-based architectures achieved breakthrough performance in natural language processing and computer vision, yet they remain inferior to simpler linear baselines in multivariate long-term forecasting. To better understand this phenomenon, we start by studying a toy linear forecasting problem for which we show that transformers are incapable of converging to their true solution despite their high expressive power. We further identify the attention of transformers as being responsible for this low generalization capacity. Building upon this insight, we propose a shallow lightweight tran"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2402.10198","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-02-15T18:55:05Z","cross_cats_sorted":["stat.ML"],"title_canon_sha256":"ac789a5e433167406f28c4fae673f33b8fe24e5b6ac4599f0208580b5e24d520","abstract_canon_sha256":"e1418635415dd7c92ac99c377e67b386742ea6ef14c1d9a5a45572e4785b1f2a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:26:19.676849Z","signature_b64":"N9bPkaX8NpOMDrvYmfUGCtdKKuyxkzCcOX+8282n0+K9hI4DUXJqqpPmBe7rFUvN4U6kJR6SzvM6XThULWriCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"7553ece058cbb541c17df9b93fe2d166bbb5ff74810f3c7fe073f5844f5a2ff7","last_reissued_at":"2026-07-05T08:26:19.676369Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:26:19.676369Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"SAMformer: Unlocking the Potential of Transformers in Time Series Forecasting with Sharpness-Aware Minimization and Channel-Wise Attention","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["stat.ML"],"primary_cat":"cs.LG","authors_text":"Aladin Virmaux, Ambroise Odonnat, Giuseppe Paolo, Ievgen Redko, Romain Ilbert, Themis Palpanas, Vasilii Feofanov","submitted_at":"2024-02-15T18:55:05Z","abstract_excerpt":"Transformer-based architectures achieved breakthrough performance in natural language processing and computer vision, yet they remain inferior to simpler linear baselines in multivariate long-term forecasting. To better understand this phenomenon, we start by studying a toy linear forecasting problem for which we show that transformers are incapable of converging to their true solution despite their high expressive power. We further identify the attention of transformers as being responsible for this low generalization capacity. Building upon this insight, we propose a shallow lightweight tran"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2402.10198","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2402.10198/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2402.10198","created_at":"2026-07-05T08:26:19.676428+00:00"},{"alias_kind":"arxiv_version","alias_value":"2402.10198v3","created_at":"2026-07-05T08:26:19.676428+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2402.10198","created_at":"2026-07-05T08:26:19.676428+00:00"},{"alias_kind":"pith_short_12","alias_value":"OVJ6ZYCYZO2U","created_at":"2026-07-05T08:26:19.676428+00:00"},{"alias_kind":"pith_short_16","alias_value":"OVJ6ZYCYZO2UDQL5","created_at":"2026-07-05T08:26:19.676428+00:00"},{"alias_kind":"pith_short_8","alias_value":"OVJ6ZYCY","created_at":"2026-07-05T08:26:19.676428+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2511.08947","citing_title":"AlphaCast: A Human Wisdom-LLM Intelligence Co-Reasoning Framework for Interactive Time Series Forecasting","ref_index":10,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/OVJ6ZYCYZO2UDQL57G4T7YWRM2","json":"https://pith.science/pith/OVJ6ZYCYZO2UDQL57G4T7YWRM2.json","graph_json":"https://pith.science/api/pith-number/OVJ6ZYCYZO2UDQL57G4T7YWRM2/graph.json","events_json":"https://pith.science/api/pith-number/OVJ6ZYCYZO2UDQL57G4T7YWRM2/events.json","paper":"https://pith.science/paper/OVJ6ZYCY"},"agent_actions":{"view_html":"https://pith.science/pith/OVJ6ZYCYZO2UDQL57G4T7YWRM2","download_json":"https://pith.science/pith/OVJ6ZYCYZO2UDQL57G4T7YWRM2.json","view_paper":"https://pith.science/paper/OVJ6ZYCY","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2402.10198&json=true","fetch_graph":"https://pith.science/api/pith-number/OVJ6ZYCYZO2UDQL57G4T7YWRM2/graph.json","fetch_events":"https://pith.science/api/pith-number/OVJ6ZYCYZO2UDQL57G4T7YWRM2/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/OVJ6ZYCYZO2UDQL57G4T7YWRM2/action/timestamp_anchor","attest_storage":"https://pith.science/pith/OVJ6ZYCYZO2UDQL57G4T7YWRM2/action/storage_attestation","attest_author":"https://pith.science/pith/OVJ6ZYCYZO2UDQL57G4T7YWRM2/action/author_attestation","sign_citation":"https://pith.science/pith/OVJ6ZYCYZO2UDQL57G4T7YWRM2/action/citation_signature","submit_replication":"https://pith.science/pith/OVJ6ZYCYZO2UDQL57G4T7YWRM2/action/replication_record"}},"created_at":"2026-07-05T08:26:19.676428+00:00","updated_at":"2026-07-05T08:26:19.676428+00:00"}