{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:YNOJFTPSYL6R4EJRD2ICGZUSEF","short_pith_number":"pith:YNOJFTPS","schema_version":"1.0","canonical_sha256":"c35c92cdf2c2fd1e11311e902366922160c4b4b3e1caa8159c7fb31ee4423920","source":{"kind":"arxiv","id":"2410.04271","version":2},"attestation_state":"computed","paper":{"title":"Fundamental Limitations on Subquadratic Alternatives to Transformers","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CC","cs.CL"],"primary_cat":"cs.LG","authors_text":"Hantao Yu, Josh Alman","submitted_at":"2024-10-05T19:21:13Z","abstract_excerpt":"The Transformer architecture is widely deployed in many popular and impactful Large Language Models. At its core is the attention mechanism for calculating correlations between pairs of tokens. Performing an attention computation takes quadratic time in the input size, and had become the time bottleneck for transformer operations. In order to circumvent this, researchers have used a variety of approaches, including designing heuristic algorithms for performing attention computations faster, and proposing alternatives to the attention mechanism which can be computed more quickly. For instance, "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.04271","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-10-05T19:21:13Z","cross_cats_sorted":["cs.CC","cs.CL"],"title_canon_sha256":"3c17a17ddaae890af003da2caeab0b2fed29d3591e651dee6d7fa4022d9bb6eb","abstract_canon_sha256":"280ffc18f3b3aabc27e3e96ba33973c32395dcd103836e7c8e559f605ac5a742"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:08:01.455649Z","signature_b64":"8nLVlxnoJHySuv6E1tYFvs4icDHimedaMwHryExxBSeCO4udexNEH4z/ENdd2ARX2jrBs8SXf613d+ws3tyBAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"c35c92cdf2c2fd1e11311e902366922160c4b4b3e1caa8159c7fb31ee4423920","last_reissued_at":"2026-07-05T11:08:01.455119Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:08:01.455119Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Fundamental Limitations on Subquadratic Alternatives to Transformers","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CC","cs.CL"],"primary_cat":"cs.LG","authors_text":"Hantao Yu, Josh Alman","submitted_at":"2024-10-05T19:21:13Z","abstract_excerpt":"The Transformer architecture is widely deployed in many popular and impactful Large Language Models. At its core is the attention mechanism for calculating correlations between pairs of tokens. Performing an attention computation takes quadratic time in the input size, and had become the time bottleneck for transformer operations. In order to circumvent this, researchers have used a variety of approaches, including designing heuristic algorithms for performing attention computations faster, and proposing alternatives to the attention mechanism which can be computed more quickly. For instance, "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.04271","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.04271/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.04271","created_at":"2026-07-05T11:08:01.455177+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.04271v2","created_at":"2026-07-05T11:08:01.455177+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.04271","created_at":"2026-07-05T11:08:01.455177+00:00"},{"alias_kind":"pith_short_12","alias_value":"YNOJFTPSYL6R","created_at":"2026-07-05T11:08:01.455177+00:00"},{"alias_kind":"pith_short_16","alias_value":"YNOJFTPSYL6R4EJR","created_at":"2026-07-05T11:08:01.455177+00:00"},{"alias_kind":"pith_short_8","alias_value":"YNOJFTPS","created_at":"2026-07-05T11:08:01.455177+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2604.17935","citing_title":"How Much Cache Does Reasoning Need? Depth-Cache Tradeoffs in KV-Compressed Transformers","ref_index":1,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/YNOJFTPSYL6R4EJRD2ICGZUSEF","json":"https://pith.science/pith/YNOJFTPSYL6R4EJRD2ICGZUSEF.json","graph_json":"https://pith.science/api/pith-number/YNOJFTPSYL6R4EJRD2ICGZUSEF/graph.json","events_json":"https://pith.science/api/pith-number/YNOJFTPSYL6R4EJRD2ICGZUSEF/events.json","paper":"https://pith.science/paper/YNOJFTPS"},"agent_actions":{"view_html":"https://pith.science/pith/YNOJFTPSYL6R4EJRD2ICGZUSEF","download_json":"https://pith.science/pith/YNOJFTPSYL6R4EJRD2ICGZUSEF.json","view_paper":"https://pith.science/paper/YNOJFTPS","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.04271&json=true","fetch_graph":"https://pith.science/api/pith-number/YNOJFTPSYL6R4EJRD2ICGZUSEF/graph.json","fetch_events":"https://pith.science/api/pith-number/YNOJFTPSYL6R4EJRD2ICGZUSEF/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/YNOJFTPSYL6R4EJRD2ICGZUSEF/action/timestamp_anchor","attest_storage":"https://pith.science/pith/YNOJFTPSYL6R4EJRD2ICGZUSEF/action/storage_attestation","attest_author":"https://pith.science/pith/YNOJFTPSYL6R4EJRD2ICGZUSEF/action/author_attestation","sign_citation":"https://pith.science/pith/YNOJFTPSYL6R4EJRD2ICGZUSEF/action/citation_signature","submit_replication":"https://pith.science/pith/YNOJFTPSYL6R4EJRD2ICGZUSEF/action/replication_record"}},"created_at":"2026-07-05T11:08:01.455177+00:00","updated_at":"2026-07-05T11:08:01.455177+00:00"}