{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:3ICSKRYLHHMDYZN647GYC7PW4U","short_pith_number":"pith:3ICSKRYL","schema_version":"1.0","canonical_sha256":"da0525470b39d83c65bee7cd817df6e516e78ee17c9852d4b32280ddc8e3a52f","source":{"kind":"arxiv","id":"2302.02451","version":2},"attestation_state":"computed","paper":{"title":"KDEformer: Accelerating Transformers via Kernel Density Estimation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CV","cs.DS"],"primary_cat":"cs.LG","authors_text":"Amin Karbasi, Amir Zandieh, Insu Han, Majid Daliri","submitted_at":"2023-02-05T18:23:49Z","abstract_excerpt":"Dot-product attention mechanism plays a crucial role in modern deep architectures (e.g., Transformer) for sequence modeling, however, na\\\"ive exact computation of this model incurs quadratic time and memory complexities in sequence length, hindering the training of long-sequence models. Critical bottlenecks are due to the computation of partition functions in the denominator of softmax function as well as the multiplication of the softmax matrix with the matrix of values. Our key observation is that the former can be reduced to a variant of the kernel density estimation (KDE) problem, and an e"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2302.02451","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2023-02-05T18:23:49Z","cross_cats_sorted":["cs.CV","cs.DS"],"title_canon_sha256":"eaec998d47c9e0de447064dac7eb8e40db169f449fae0bfae65385a050774350","abstract_canon_sha256":"f25ea5e62131980a9fd548c189db40e25d6e388754a894c2f08425c88ef493b8"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:26:00.573747Z","signature_b64":"0oUzUrBIx/8PrBf7NjVCSfD0O83RuLtVxdKSKZ2DfqY0gAWMif/hGw3pzINfecb46+e/v+G8ksUW9TF1gb5ACg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"da0525470b39d83c65bee7cd817df6e516e78ee17c9852d4b32280ddc8e3a52f","last_reissued_at":"2026-07-05T06:26:00.573321Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:26:00.573321Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"KDEformer: Accelerating Transformers via Kernel Density Estimation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CV","cs.DS"],"primary_cat":"cs.LG","authors_text":"Amin Karbasi, Amir Zandieh, Insu Han, Majid Daliri","submitted_at":"2023-02-05T18:23:49Z","abstract_excerpt":"Dot-product attention mechanism plays a crucial role in modern deep architectures (e.g., Transformer) for sequence modeling, however, na\\\"ive exact computation of this model incurs quadratic time and memory complexities in sequence length, hindering the training of long-sequence models. Critical bottlenecks are due to the computation of partition functions in the denominator of softmax function as well as the multiplication of the softmax matrix with the matrix of values. Our key observation is that the former can be reduced to a variant of the kernel density estimation (KDE) problem, and an e"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2302.02451","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2302.02451/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2302.02451","created_at":"2026-07-05T06:26:00.573377+00:00"},{"alias_kind":"arxiv_version","alias_value":"2302.02451v2","created_at":"2026-07-05T06:26:00.573377+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2302.02451","created_at":"2026-07-05T06:26:00.573377+00:00"},{"alias_kind":"pith_short_12","alias_value":"3ICSKRYLHHMD","created_at":"2026-07-05T06:26:00.573377+00:00"},{"alias_kind":"pith_short_16","alias_value":"3ICSKRYLHHMDYZN6","created_at":"2026-07-05T06:26:00.573377+00:00"},{"alias_kind":"pith_short_8","alias_value":"3ICSKRYL","created_at":"2026-07-05T06:26:00.573377+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2306.14048","citing_title":"H$_2$O: Heavy-Hitter Oracle for Efficient Generative Inference of Large Language Models","ref_index":97,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/3ICSKRYLHHMDYZN647GYC7PW4U","json":"https://pith.science/pith/3ICSKRYLHHMDYZN647GYC7PW4U.json","graph_json":"https://pith.science/api/pith-number/3ICSKRYLHHMDYZN647GYC7PW4U/graph.json","events_json":"https://pith.science/api/pith-number/3ICSKRYLHHMDYZN647GYC7PW4U/events.json","paper":"https://pith.science/paper/3ICSKRYL"},"agent_actions":{"view_html":"https://pith.science/pith/3ICSKRYLHHMDYZN647GYC7PW4U","download_json":"https://pith.science/pith/3ICSKRYLHHMDYZN647GYC7PW4U.json","view_paper":"https://pith.science/paper/3ICSKRYL","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2302.02451&json=true","fetch_graph":"https://pith.science/api/pith-number/3ICSKRYLHHMDYZN647GYC7PW4U/graph.json","fetch_events":"https://pith.science/api/pith-number/3ICSKRYLHHMDYZN647GYC7PW4U/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/3ICSKRYLHHMDYZN647GYC7PW4U/action/timestamp_anchor","attest_storage":"https://pith.science/pith/3ICSKRYLHHMDYZN647GYC7PW4U/action/storage_attestation","attest_author":"https://pith.science/pith/3ICSKRYLHHMDYZN647GYC7PW4U/action/author_attestation","sign_citation":"https://pith.science/pith/3ICSKRYLHHMDYZN647GYC7PW4U/action/citation_signature","submit_replication":"https://pith.science/pith/3ICSKRYLHHMDYZN647GYC7PW4U/action/replication_record"}},"created_at":"2026-07-05T06:26:00.573377+00:00","updated_at":"2026-07-05T06:26:00.573377+00:00"}