{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:FLFGC45XB6AKFAGBNJPQLLTFSP","short_pith_number":"pith:FLFGC45X","schema_version":"1.0","canonical_sha256":"2aca6173b70f80a280c16a5f05ae6593d926bf88fe3488b1e967673e202c743b","source":{"kind":"arxiv","id":"2312.04985","version":6},"attestation_state":"computed","paper":{"title":"SparQ Attention: Bandwidth-Efficient LLM Inference","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Carlo Luschi, Charlie Blake, Douglas Orr, Ivan Chelombiev, Luka Ribar, Luke Hudlass-Galley","submitted_at":"2023-12-08T11:47:35Z","abstract_excerpt":"The computational difficulties of large language model (LLM) inference remain a significant obstacle to their widespread deployment. The need for many applications to support long input sequences and process them in large batches typically causes token-generation to be bottlenecked by data transfer. For this reason, we introduce SparQ Attention, a technique for increasing the inference throughput of LLMs by utilising memory bandwidth more efficiently within the attention layers, through selective fetching of the cached history. Our proposed technique can be applied directly to off-the-shelf LL"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2312.04985","kind":"arxiv","version":6},"metadata":{"license":"http://creativecommons.org/licenses/by-sa/4.0/","primary_cat":"cs.LG","submitted_at":"2023-12-08T11:47:35Z","cross_cats_sorted":[],"title_canon_sha256":"5373ef6f62c02b9dc28e405855e269b1a894523cdea6d52e40ad50eb247e018c","abstract_canon_sha256":"af11020fe59d8691060030d81d19998bc28183c8c142657f21c3f61c7db256f7"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:02:55.499611Z","signature_b64":"S8ic590DADUwl2rGlWu6n3L+2stDsjLWHqBQQSaxm31znurdECILO5wGJs+Y/coH29G5YISTkqZrLR47F+j6BA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"2aca6173b70f80a280c16a5f05ae6593d926bf88fe3488b1e967673e202c743b","last_reissued_at":"2026-07-05T09:02:55.499088Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:02:55.499088Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"SparQ Attention: Bandwidth-Efficient LLM Inference","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Carlo Luschi, Charlie Blake, Douglas Orr, Ivan Chelombiev, Luka Ribar, Luke Hudlass-Galley","submitted_at":"2023-12-08T11:47:35Z","abstract_excerpt":"The computational difficulties of large language model (LLM) inference remain a significant obstacle to their widespread deployment. The need for many applications to support long input sequences and process them in large batches typically causes token-generation to be bottlenecked by data transfer. For this reason, we introduce SparQ Attention, a technique for increasing the inference throughput of LLMs by utilising memory bandwidth more efficiently within the attention layers, through selective fetching of the cached history. Our proposed technique can be applied directly to off-the-shelf LL"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2312.04985","kind":"arxiv","version":6},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2312.04985/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2312.04985","created_at":"2026-07-05T09:02:55.499152+00:00"},{"alias_kind":"arxiv_version","alias_value":"2312.04985v6","created_at":"2026-07-05T09:02:55.499152+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2312.04985","created_at":"2026-07-05T09:02:55.499152+00:00"},{"alias_kind":"pith_short_12","alias_value":"FLFGC45XB6AK","created_at":"2026-07-05T09:02:55.499152+00:00"},{"alias_kind":"pith_short_16","alias_value":"FLFGC45XB6AKFAGB","created_at":"2026-07-05T09:02:55.499152+00:00"},{"alias_kind":"pith_short_8","alias_value":"FLFGC45X","created_at":"2026-07-05T09:02:55.499152+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2505.05772","citing_title":"Sparse Attention Remapping with Clustering for Efficient LLM Decoding on PIM","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2507.21433","citing_title":"ReasonCache: Accelerating Large Reasoning Model Serving through KV Cache Sharing","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2512.16056","citing_title":"MultiPath Memory Access: Breaking Host-GPU Bandwidth Bottlenecks in LLM Services","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2604.24820","citing_title":"Salca: A Sparsity-Aware Hardware Accelerator for Efficient Long-Context Attention Decoding","ref_index":44,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07719","citing_title":"An Efficient Hybrid Sparse Attention with CPU-GPU Parallelism for Long-Context Inference","ref_index":38,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06763","citing_title":"Sparse Attention as a Range Searching Problem: Towards an Inference-Efficient Index for KV Cache","ref_index":39,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/FLFGC45XB6AKFAGBNJPQLLTFSP","json":"https://pith.science/pith/FLFGC45XB6AKFAGBNJPQLLTFSP.json","graph_json":"https://pith.science/api/pith-number/FLFGC45XB6AKFAGBNJPQLLTFSP/graph.json","events_json":"https://pith.science/api/pith-number/FLFGC45XB6AKFAGBNJPQLLTFSP/events.json","paper":"https://pith.science/paper/FLFGC45X"},"agent_actions":{"view_html":"https://pith.science/pith/FLFGC45XB6AKFAGBNJPQLLTFSP","download_json":"https://pith.science/pith/FLFGC45XB6AKFAGBNJPQLLTFSP.json","view_paper":"https://pith.science/paper/FLFGC45X","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2312.04985&json=true","fetch_graph":"https://pith.science/api/pith-number/FLFGC45XB6AKFAGBNJPQLLTFSP/graph.json","fetch_events":"https://pith.science/api/pith-number/FLFGC45XB6AKFAGBNJPQLLTFSP/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/FLFGC45XB6AKFAGBNJPQLLTFSP/action/timestamp_anchor","attest_storage":"https://pith.science/pith/FLFGC45XB6AKFAGBNJPQLLTFSP/action/storage_attestation","attest_author":"https://pith.science/pith/FLFGC45XB6AKFAGBNJPQLLTFSP/action/author_attestation","sign_citation":"https://pith.science/pith/FLFGC45XB6AKFAGBNJPQLLTFSP/action/citation_signature","submit_replication":"https://pith.science/pith/FLFGC45XB6AKFAGBNJPQLLTFSP/action/replication_record"}},"created_at":"2026-07-05T09:02:55.499152+00:00","updated_at":"2026-07-05T09:02:55.499152+00:00"}