{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:5TSRCTMN3YHRMTRGN7YV3EJCH7","short_pith_number":"pith:5TSRCTMN","schema_version":"1.0","canonical_sha256":"ece5114d8dde0f164e266ff15d91223fe57bfd0ea94a4214798a99afd4153b09","source":{"kind":"arxiv","id":"2410.16179","version":4},"attestation_state":"computed","paper":{"title":"MagicPIG: LSH Sampling for Efficient LLM Generation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Beidi Chen, Jianyu Zhang, Leon Bottou, Matthijs Douze, Niklas Nolte, Ranajoy Sadhukhan, Yang Zhou, Yuandong Tian, Zhihao Jia, Zhuoming Chen, Zihao Ye","submitted_at":"2024-10-21T16:44:51Z","abstract_excerpt":"Large language models (LLMs) with long context windows have gained significant attention. However, the KV cache, stored to avoid re-computation, becomes a bottleneck. Various dynamic sparse or TopK-based attention approximation methods have been proposed to leverage the common insight that attention is sparse. In this paper, we first show that TopK attention itself suffers from quality degradation in certain downstream tasks because attention is not always as sparse as expected. Rather than selecting the keys and values with the highest attention scores, sampling with theoretical guarantees ca"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.16179","kind":"arxiv","version":4},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-10-21T16:44:51Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"b8ea6f1e8cc77f8377660410c7862d8c91f66c6fc9d84d7a95272176d1fd7609","abstract_canon_sha256":"d15b784de3d0d5d1947ede8f542abdfad7cb56e271bbd58ccb6d5efedd5567b2"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:50:53.257588Z","signature_b64":"5JpdiOdywmz9Gcq5owBnxRCzeTPntNzImogQTvgqKab6hQw3kf+J2eXO/Q/PPUfbKrdNTS8WAqt4oH/grHQXCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ece5114d8dde0f164e266ff15d91223fe57bfd0ea94a4214798a99afd4153b09","last_reissued_at":"2026-07-05T09:50:53.257080Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:50:53.257080Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"MagicPIG: LSH Sampling for Efficient LLM Generation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Beidi Chen, Jianyu Zhang, Leon Bottou, Matthijs Douze, Niklas Nolte, Ranajoy Sadhukhan, Yang Zhou, Yuandong Tian, Zhihao Jia, Zhuoming Chen, Zihao Ye","submitted_at":"2024-10-21T16:44:51Z","abstract_excerpt":"Large language models (LLMs) with long context windows have gained significant attention. However, the KV cache, stored to avoid re-computation, becomes a bottleneck. Various dynamic sparse or TopK-based attention approximation methods have been proposed to leverage the common insight that attention is sparse. In this paper, we first show that TopK attention itself suffers from quality degradation in certain downstream tasks because attention is not always as sparse as expected. Rather than selecting the keys and values with the highest attention scores, sampling with theoretical guarantees ca"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.16179","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.16179/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.16179","created_at":"2026-07-05T09:50:53.257134+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.16179v4","created_at":"2026-07-05T09:50:53.257134+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.16179","created_at":"2026-07-05T09:50:53.257134+00:00"},{"alias_kind":"pith_short_12","alias_value":"5TSRCTMN3YHR","created_at":"2026-07-05T09:50:53.257134+00:00"},{"alias_kind":"pith_short_16","alias_value":"5TSRCTMN3YHRMTRG","created_at":"2026-07-05T09:50:53.257134+00:00"},{"alias_kind":"pith_short_8","alias_value":"5TSRCTMN","created_at":"2026-07-05T09:50:53.257134+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":12,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.06453","citing_title":"Vortex: Efficient and Programmable Sparse Attention Serving for AI Agents","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2605.21649","citing_title":"EntmaxKV: Support-Aware Decoding for Entmax Attention","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2502.11089","citing_title":"Native Sparse Attention: Hardware-Aligned and Natively Trainable Sparse Attention","ref_index":85,"is_internal_anchor":false},{"citing_arxiv_id":"2604.08584","citing_title":"CSAttention: Centroid-Scoring Attention for Accelerating LLM Inference","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12110","citing_title":"AB-Sparse: Sparse Attention with Adaptive Block Size for Accurate and Efficient Long-Context Inference","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2406.02069","citing_title":"PyramidKV: Dynamic KV Cache Compression based on Pyramidal Information Funneling","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2401.08281","citing_title":"The Faiss library","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2604.10539","citing_title":"IceCache: Memory-efficient KV-cache Management for Long-Sequence LLMs","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07363","citing_title":"MISA: Mixture of Indexer Sparse Attention for Long-Context LLM Inference","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07719","citing_title":"An Efficient Hybrid Sparse Attention with CPU-GPU Parallelism for Long-Context Inference","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06763","citing_title":"Sparse Attention as a Range Searching Problem: Towards an Inference-Efficient Index for KV Cache","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2604.16983","citing_title":"Graph-Guided Adaptive Channel Elimination for KV Cache Compression","ref_index":16,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/5TSRCTMN3YHRMTRGN7YV3EJCH7","json":"https://pith.science/pith/5TSRCTMN3YHRMTRGN7YV3EJCH7.json","graph_json":"https://pith.science/api/pith-number/5TSRCTMN3YHRMTRGN7YV3EJCH7/graph.json","events_json":"https://pith.science/api/pith-number/5TSRCTMN3YHRMTRGN7YV3EJCH7/events.json","paper":"https://pith.science/paper/5TSRCTMN"},"agent_actions":{"view_html":"https://pith.science/pith/5TSRCTMN3YHRMTRGN7YV3EJCH7","download_json":"https://pith.science/pith/5TSRCTMN3YHRMTRGN7YV3EJCH7.json","view_paper":"https://pith.science/paper/5TSRCTMN","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.16179&json=true","fetch_graph":"https://pith.science/api/pith-number/5TSRCTMN3YHRMTRGN7YV3EJCH7/graph.json","fetch_events":"https://pith.science/api/pith-number/5TSRCTMN3YHRMTRGN7YV3EJCH7/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/5TSRCTMN3YHRMTRGN7YV3EJCH7/action/timestamp_anchor","attest_storage":"https://pith.science/pith/5TSRCTMN3YHRMTRGN7YV3EJCH7/action/storage_attestation","attest_author":"https://pith.science/pith/5TSRCTMN3YHRMTRGN7YV3EJCH7/action/author_attestation","sign_citation":"https://pith.science/pith/5TSRCTMN3YHRMTRGN7YV3EJCH7/action/citation_signature","submit_replication":"https://pith.science/pith/5TSRCTMN3YHRMTRGN7YV3EJCH7/action/replication_record"}},"created_at":"2026-07-05T09:50:53.257134+00:00","updated_at":"2026-07-05T09:50:53.257134+00:00"}