{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:PSKI4TU4NHQWPGGAQD6XGE3PA2","short_pith_number":"pith:PSKI4TU4","schema_version":"1.0","canonical_sha256":"7c948e4e9c69e16798c080fd73136f0693191cdb7e4dcd38695e62075d160d58","source":{"kind":"arxiv","id":"2310.01655","version":3},"attestation_state":"computed","paper":{"title":"PolySketchFormer: Fast Transformers via Sketching Polynomial Kernels","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Peilin Zhong, Praneeth Kacham, Vahab Mirrokni","submitted_at":"2023-10-02T21:39:04Z","abstract_excerpt":"The quadratic time and memory complexity inherent to self-attention mechanisms, with respect to sequence length, presents a critical computational bottleneck in the training and deployment of large-scale Transformer-based language models. Recent theoretical results indicate the intractability of sub-quadratic softmax attention approximation under reasonable complexity assumptions. This paper addresses this challenge by first demonstrating that polynomial attention with high degree can effectively replace softmax without sacrificing model quality. Next, we develop polynomial sketching technique"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2310.01655","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2023-10-02T21:39:04Z","cross_cats_sorted":[],"title_canon_sha256":"d9b91e09fd3840046cfa0e9eeba7fe3905304f69d9bc8254f3012ad2737a73b9","abstract_canon_sha256":"14104c9f8d99ae52dc5ff4b033992746aa7e36e363b94ed599aa0ba2b76eef5a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:56:53.265894Z","signature_b64":"K0Nve9SvwdRfr6KQ1wCEXAlMJ18Zkd62A0J5kZcLKjU0n9oQdVZx+E7Cg3VQMfBn+sFYsf+h6Uivds2A14bWAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"7c948e4e9c69e16798c080fd73136f0693191cdb7e4dcd38695e62075d160d58","last_reissued_at":"2026-07-05T07:56:53.265427Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:56:53.265427Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"PolySketchFormer: Fast Transformers via Sketching Polynomial Kernels","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Peilin Zhong, Praneeth Kacham, Vahab Mirrokni","submitted_at":"2023-10-02T21:39:04Z","abstract_excerpt":"The quadratic time and memory complexity inherent to self-attention mechanisms, with respect to sequence length, presents a critical computational bottleneck in the training and deployment of large-scale Transformer-based language models. Recent theoretical results indicate the intractability of sub-quadratic softmax attention approximation under reasonable complexity assumptions. This paper addresses this challenge by first demonstrating that polynomial attention with high degree can effectively replace softmax without sacrificing model quality. Next, we develop polynomial sketching technique"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2310.01655","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2310.01655/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2310.01655","created_at":"2026-07-05T07:56:53.265485+00:00"},{"alias_kind":"arxiv_version","alias_value":"2310.01655v3","created_at":"2026-07-05T07:56:53.265485+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2310.01655","created_at":"2026-07-05T07:56:53.265485+00:00"},{"alias_kind":"pith_short_12","alias_value":"PSKI4TU4NHQW","created_at":"2026-07-05T07:56:53.265485+00:00"},{"alias_kind":"pith_short_16","alias_value":"PSKI4TU4NHQWPGGA","created_at":"2026-07-05T07:56:53.265485+00:00"},{"alias_kind":"pith_short_8","alias_value":"PSKI4TU4","created_at":"2026-07-05T07:56:53.265485+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.03825","citing_title":"Dynamic Short Convolutions Improve Transformers","ref_index":72,"is_internal_anchor":false},{"citing_arxiv_id":"2511.17388","citing_title":"Selective Rotary Position Embedding","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2306.14048","citing_title":"H$_2$O: Heavy-Hitter Oracle for Efficient Generative Inference of Large Language Models","ref_index":106,"is_internal_anchor":false},{"citing_arxiv_id":"2404.14294","citing_title":"A Survey on Efficient Inference for Large Language Models","ref_index":87,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/PSKI4TU4NHQWPGGAQD6XGE3PA2","json":"https://pith.science/pith/PSKI4TU4NHQWPGGAQD6XGE3PA2.json","graph_json":"https://pith.science/api/pith-number/PSKI4TU4NHQWPGGAQD6XGE3PA2/graph.json","events_json":"https://pith.science/api/pith-number/PSKI4TU4NHQWPGGAQD6XGE3PA2/events.json","paper":"https://pith.science/paper/PSKI4TU4"},"agent_actions":{"view_html":"https://pith.science/pith/PSKI4TU4NHQWPGGAQD6XGE3PA2","download_json":"https://pith.science/pith/PSKI4TU4NHQWPGGAQD6XGE3PA2.json","view_paper":"https://pith.science/paper/PSKI4TU4","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2310.01655&json=true","fetch_graph":"https://pith.science/api/pith-number/PSKI4TU4NHQWPGGAQD6XGE3PA2/graph.json","fetch_events":"https://pith.science/api/pith-number/PSKI4TU4NHQWPGGAQD6XGE3PA2/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/PSKI4TU4NHQWPGGAQD6XGE3PA2/action/timestamp_anchor","attest_storage":"https://pith.science/pith/PSKI4TU4NHQWPGGAQD6XGE3PA2/action/storage_attestation","attest_author":"https://pith.science/pith/PSKI4TU4NHQWPGGAQD6XGE3PA2/action/author_attestation","sign_citation":"https://pith.science/pith/PSKI4TU4NHQWPGGAQD6XGE3PA2/action/citation_signature","submit_replication":"https://pith.science/pith/PSKI4TU4NHQWPGGAQD6XGE3PA2/action/replication_record"}},"created_at":"2026-07-05T07:56:53.265485+00:00","updated_at":"2026-07-05T07:56:53.265485+00:00"}