{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:HWHR52P5VMGU45S23DEQGLBYFJ","short_pith_number":"pith:HWHR52P5","schema_version":"1.0","canonical_sha256":"3d8f1ee9fdab0d4e765ad8c9032c382a7e34e4a1938f8ad26b961bebbea65a45","source":{"kind":"arxiv","id":"2406.15486","version":3},"attestation_state":"computed","paper":{"title":"SampleAttention: Near-Lossless Acceleration of Long Context LLM Inference with Adaptive Structured Sparse Attention","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Chang Chen, Chao Yang, Dahua Lin, Guanyu Feng, Jiangfei Duan, Qianchao Zhu, Siran Liu, Xiao Chuanfu, Xin Lv","submitted_at":"2024-06-17T11:05:15Z","abstract_excerpt":"Large language models (LLMs) now support extremely long context windows, but the quadratic complexity of vanilla attention results in significantly long Time-to-First-Token (TTFT) latency. Existing approaches to address this complexity require additional pretraining or finetuning, and often sacrifice model accuracy. In this paper, we first provide both theoretical and empirical foundations for near-lossless sparse attention. We find dynamically capturing head-specific sparse patterns at runtime with low overhead is crucial. To address this, we propose SampleAttention, an adaptive structured an"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2406.15486","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-06-17T11:05:15Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"658adb9838894b1ef36ee5c814a4ba0d628b9ba150beee090dbdee5ddf458444","abstract_canon_sha256":"ac78ddf832f5985ad454f5b68668722c19775c7e1f3fd66af8e69b346edb25f0"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T12:03:42.883226Z","signature_b64":"rEJtqFM8jUAVfERNhMb3G6UP9WvvJsF7idakischXnqod+jxH0+1G6hgf0KesfiCVs/CvkxrAvssGbMxRTlrAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"3d8f1ee9fdab0d4e765ad8c9032c382a7e34e4a1938f8ad26b961bebbea65a45","last_reissued_at":"2026-07-05T12:03:42.882603Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T12:03:42.882603Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"SampleAttention: Near-Lossless Acceleration of Long Context LLM Inference with Adaptive Structured Sparse Attention","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Chang Chen, Chao Yang, Dahua Lin, Guanyu Feng, Jiangfei Duan, Qianchao Zhu, Siran Liu, Xiao Chuanfu, Xin Lv","submitted_at":"2024-06-17T11:05:15Z","abstract_excerpt":"Large language models (LLMs) now support extremely long context windows, but the quadratic complexity of vanilla attention results in significantly long Time-to-First-Token (TTFT) latency. Existing approaches to address this complexity require additional pretraining or finetuning, and often sacrifice model accuracy. In this paper, we first provide both theoretical and empirical foundations for near-lossless sparse attention. We find dynamically capturing head-specific sparse patterns at runtime with low overhead is crucial. To address this, we propose SampleAttention, an adaptive structured an"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.15486","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.15486/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2406.15486","created_at":"2026-07-05T12:03:42.882688+00:00"},{"alias_kind":"arxiv_version","alias_value":"2406.15486v3","created_at":"2026-07-05T12:03:42.882688+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.15486","created_at":"2026-07-05T12:03:42.882688+00:00"},{"alias_kind":"pith_short_12","alias_value":"HWHR52P5VMGU","created_at":"2026-07-05T12:03:42.882688+00:00"},{"alias_kind":"pith_short_16","alias_value":"HWHR52P5VMGU45S2","created_at":"2026-07-05T12:03:42.882688+00:00"},{"alias_kind":"pith_short_8","alias_value":"HWHR52P5","created_at":"2026-07-05T12:03:42.882688+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.09441","citing_title":"SIFT: Selective-Index For Fast Compute of RAG Prefill by Exploiting Attention Invariance","ref_index":40,"is_internal_anchor":false},{"citing_arxiv_id":"2605.00831","citing_title":"GhostServe: A Lightweight Checkpointing System in the Shadow for Fault-Tolerant LLM Serving","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2604.08584","citing_title":"CSAttention: Centroid-Scoring Attention for Accelerating LLM Inference","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06763","citing_title":"Sparse Attention as a Range Searching Problem: Towards an Inference-Efficient Index for KV Cache","ref_index":57,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/HWHR52P5VMGU45S23DEQGLBYFJ","json":"https://pith.science/pith/HWHR52P5VMGU45S23DEQGLBYFJ.json","graph_json":"https://pith.science/api/pith-number/HWHR52P5VMGU45S23DEQGLBYFJ/graph.json","events_json":"https://pith.science/api/pith-number/HWHR52P5VMGU45S23DEQGLBYFJ/events.json","paper":"https://pith.science/paper/HWHR52P5"},"agent_actions":{"view_html":"https://pith.science/pith/HWHR52P5VMGU45S23DEQGLBYFJ","download_json":"https://pith.science/pith/HWHR52P5VMGU45S23DEQGLBYFJ.json","view_paper":"https://pith.science/paper/HWHR52P5","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2406.15486&json=true","fetch_graph":"https://pith.science/api/pith-number/HWHR52P5VMGU45S23DEQGLBYFJ/graph.json","fetch_events":"https://pith.science/api/pith-number/HWHR52P5VMGU45S23DEQGLBYFJ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/HWHR52P5VMGU45S23DEQGLBYFJ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/HWHR52P5VMGU45S23DEQGLBYFJ/action/storage_attestation","attest_author":"https://pith.science/pith/HWHR52P5VMGU45S23DEQGLBYFJ/action/author_attestation","sign_citation":"https://pith.science/pith/HWHR52P5VMGU45S23DEQGLBYFJ/action/citation_signature","submit_replication":"https://pith.science/pith/HWHR52P5VMGU45S23DEQGLBYFJ/action/replication_record"}},"created_at":"2026-07-05T12:03:42.882688+00:00","updated_at":"2026-07-05T12:03:42.882688+00:00"}