{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:UMBZMPIBNHC4B625LPCDPEWPHU","short_pith_number":"pith:UMBZMPIB","schema_version":"1.0","canonical_sha256":"a303963d0169c5c0fb5d5bc43792cf3d3d56c0bb3c1d4823416733089753759a","source":{"kind":"arxiv","id":"2510.05364","version":1},"attestation_state":"computed","paper":{"title":"The End of Transformers? On Challenging Attention and the Rise of Sub-Quadratic Architectures","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Alexander M. Fichtl, Edoardo Mosca, Georg Groh, Jeremias Bohn, Josefin Kelber","submitted_at":"2025-10-06T20:45:34Z","abstract_excerpt":"Transformers have dominated sequence processing tasks for the past seven years -- most notably language modeling. However, the inherent quadratic complexity of their attention mechanism remains a significant bottleneck as context length increases. This paper surveys recent efforts to overcome this bottleneck, including advances in (sub-quadratic) attention variants, recurrent neural networks, state space models, and hybrid architectures. We critically analyze these approaches in terms of compute and memory complexity, benchmark results, and fundamental limitations to assess whether the dominan"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2510.05364","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-sa/4.0/","primary_cat":"cs.CL","submitted_at":"2025-10-06T20:45:34Z","cross_cats_sorted":[],"title_canon_sha256":"c4c2dec2929f03bc3d195740e35e1057ea0b4087319d6dcd760a5523ea4f398b","abstract_canon_sha256":"761efd82d0fa1c2a1b95d1c3e50dcbf0399348049007d6e2728ee3c64da6fe8e"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-08-18T02:15:39.294469Z","signature_b64":"xfFHPrTRf+RTxSqHq0ldJxHLuD/jgagm1HnprTJfhPvkpzpKqTDXosQkiYa3AsQ466eG83IF8eK2jeWQhxK2BQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"a303963d0169c5c0fb5d5bc43792cf3d3d56c0bb3c1d4823416733089753759a","last_reissued_at":"2026-08-18T02:15:39.292763Z","signature_status":"signed_v1","first_computed_at":"2026-08-18T02:15:39.292763Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"The End of Transformers? On Challenging Attention and the Rise of Sub-Quadratic Architectures","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Alexander M. Fichtl, Edoardo Mosca, Georg Groh, Jeremias Bohn, Josefin Kelber","submitted_at":"2025-10-06T20:45:34Z","abstract_excerpt":"Transformers have dominated sequence processing tasks for the past seven years -- most notably language modeling. However, the inherent quadratic complexity of their attention mechanism remains a significant bottleneck as context length increases. This paper surveys recent efforts to overcome this bottleneck, including advances in (sub-quadratic) attention variants, recurrent neural networks, state space models, and hybrid architectures. We critically analyze these approaches in terms of compute and memory complexity, benchmark results, and fundamental limitations to assess whether the dominan"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2510.05364","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2510.05364/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2510.05364","created_at":"2026-08-18T02:15:39.292344+00:00"},{"alias_kind":"arxiv_version","alias_value":"2510.05364v1","created_at":"2026-08-18T02:15:39.292344+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2510.05364","created_at":"2026-08-18T02:15:39.292344+00:00"},{"alias_kind":"pith_short_12","alias_value":"UMBZMPIBNHC4","created_at":"2026-08-18T02:15:39.292344+00:00"},{"alias_kind":"pith_short_16","alias_value":"UMBZMPIBNHC4B625","created_at":"2026-08-18T02:15:39.292344+00:00"},{"alias_kind":"pith_short_8","alias_value":"UMBZMPIB","created_at":"2026-08-18T02:15:39.292344+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":4,"sample":[{"citing_arxiv_id":"2605.15216","citing_title":"Hardware-Software Co-Design of Scalable, Energy-Efficient Analog Recurrent Computations","ref_index":12,"is_internal_anchor":true},{"citing_arxiv_id":"2606.12364","citing_title":"On Subquadratic Architectures: From Applications to Principles","ref_index":68,"is_internal_anchor":true},{"citing_arxiv_id":"2605.15216","citing_title":"Hardware-Software Co-Design of Scalable, Energy-Efficient Analog Recurrent Computations","ref_index":12,"is_internal_anchor":true},{"citing_arxiv_id":"2604.21454","citing_title":"Reasoning Primitives in Hybrid and Non-Hybrid LLMs: Do Architectural Differences Yield Advantages in State-Tracking and Recall?","ref_index":11,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/UMBZMPIBNHC4B625LPCDPEWPHU","json":"https://pith.science/pith/UMBZMPIBNHC4B625LPCDPEWPHU.json","graph_json":"https://pith.science/api/pith-number/UMBZMPIBNHC4B625LPCDPEWPHU/graph.json","events_json":"https://pith.science/api/pith-number/UMBZMPIBNHC4B625LPCDPEWPHU/events.json","paper":"https://pith.science/paper/UMBZMPIB"},"agent_actions":{"view_html":"https://pith.science/pith/UMBZMPIBNHC4B625LPCDPEWPHU","download_json":"https://pith.science/pith/UMBZMPIBNHC4B625LPCDPEWPHU.json","view_paper":"https://pith.science/paper/UMBZMPIB","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2510.05364&json=true","fetch_graph":"https://pith.science/api/pith-number/UMBZMPIBNHC4B625LPCDPEWPHU/graph.json","fetch_events":"https://pith.science/api/pith-number/UMBZMPIBNHC4B625LPCDPEWPHU/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/UMBZMPIBNHC4B625LPCDPEWPHU/action/timestamp_anchor","attest_storage":"https://pith.science/pith/UMBZMPIBNHC4B625LPCDPEWPHU/action/storage_attestation","attest_author":"https://pith.science/pith/UMBZMPIBNHC4B625LPCDPEWPHU/action/author_attestation","sign_citation":"https://pith.science/pith/UMBZMPIBNHC4B625LPCDPEWPHU/action/citation_signature","submit_replication":"https://pith.science/pith/UMBZMPIBNHC4B625LPCDPEWPHU/action/replication_record"}},"created_at":"2026-08-18T02:15:39.292344+00:00","updated_at":"2026-08-18T02:15:39.292344+00:00"}