{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:OT6KG5FYO5ANRD3LHWAOSZ4SJR","short_pith_number":"pith:OT6KG5FY","schema_version":"1.0","canonical_sha256":"74fca374b87740d88f6b3d80e967924c4552ed109f6f0fdb6edb4152e3222201","source":{"kind":"arxiv","id":"2506.15704","version":1},"attestation_state":"computed","paper":{"title":"Learn from the Past: Fast Sparse Indexing for Large Language Model Decoding","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Feiyu Yao, Qian Wang","submitted_at":"2025-05-30T02:35:59Z","abstract_excerpt":"As large language models (LLMs) continue to support increasingly longer contexts, the memory demand for key-value (KV) caches during decoding grows rapidly, becoming a critical bottleneck in both GPU memory capacity and PCIe bandwidth. Sparse attention mechanisms alleviate this issue by computing attention weights only for selected key-value pairs. However, their indexing computation typically requires traversing all key vectors, resulting in significant computational and data transfer overhead. To reduce the cost of index retrieval, existing methods often treat each decoding step as an indepe"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2506.15704","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-05-30T02:35:59Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"5988fff61caa31a5b25ed15b88e8b0c1ebac9ab0c3f78d96ba8fc0a14c247110","abstract_canon_sha256":"53b2b0b42cda36e81b4651434fb5720e45cc29cf2f8da041ca11a01877591d23"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:23:50.702965Z","signature_b64":"gJ6ZbttOBWIGPDjIwakwC9QIKekaMF6HG/d6jX4KRzEGd3O7WXTE1W8vRoAZbo21SLegH0j2ILFry2DO5l1RDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"74fca374b87740d88f6b3d80e967924c4552ed109f6f0fdb6edb4152e3222201","last_reissued_at":"2026-07-05T11:23:50.702458Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:23:50.702458Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Learn from the Past: Fast Sparse Indexing for Large Language Model Decoding","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Feiyu Yao, Qian Wang","submitted_at":"2025-05-30T02:35:59Z","abstract_excerpt":"As large language models (LLMs) continue to support increasingly longer contexts, the memory demand for key-value (KV) caches during decoding grows rapidly, becoming a critical bottleneck in both GPU memory capacity and PCIe bandwidth. Sparse attention mechanisms alleviate this issue by computing attention weights only for selected key-value pairs. However, their indexing computation typically requires traversing all key vectors, resulting in significant computational and data transfer overhead. To reduce the cost of index retrieval, existing methods often treat each decoding step as an indepe"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.15704","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2506.15704/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2506.15704","created_at":"2026-07-05T11:23:50.702528+00:00"},{"alias_kind":"arxiv_version","alias_value":"2506.15704v1","created_at":"2026-07-05T11:23:50.702528+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.15704","created_at":"2026-07-05T11:23:50.702528+00:00"},{"alias_kind":"pith_short_12","alias_value":"OT6KG5FYO5AN","created_at":"2026-07-05T11:23:50.702528+00:00"},{"alias_kind":"pith_short_16","alias_value":"OT6KG5FYO5ANRD3L","created_at":"2026-07-05T11:23:50.702528+00:00"},{"alias_kind":"pith_short_8","alias_value":"OT6KG5FY","created_at":"2026-07-05T11:23:50.702528+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.30389","citing_title":"Predict, Reuse, and Repair: Accelerating Dynamic Sparse Attention for Long-Context LLM Decoding","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2604.10098","citing_title":"Attention Sink in Transformers: A Survey on Utilization, Interpretation, and Mitigation","ref_index":74,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/OT6KG5FYO5ANRD3LHWAOSZ4SJR","json":"https://pith.science/pith/OT6KG5FYO5ANRD3LHWAOSZ4SJR.json","graph_json":"https://pith.science/api/pith-number/OT6KG5FYO5ANRD3LHWAOSZ4SJR/graph.json","events_json":"https://pith.science/api/pith-number/OT6KG5FYO5ANRD3LHWAOSZ4SJR/events.json","paper":"https://pith.science/paper/OT6KG5FY"},"agent_actions":{"view_html":"https://pith.science/pith/OT6KG5FYO5ANRD3LHWAOSZ4SJR","download_json":"https://pith.science/pith/OT6KG5FYO5ANRD3LHWAOSZ4SJR.json","view_paper":"https://pith.science/paper/OT6KG5FY","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2506.15704&json=true","fetch_graph":"https://pith.science/api/pith-number/OT6KG5FYO5ANRD3LHWAOSZ4SJR/graph.json","fetch_events":"https://pith.science/api/pith-number/OT6KG5FYO5ANRD3LHWAOSZ4SJR/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/OT6KG5FYO5ANRD3LHWAOSZ4SJR/action/timestamp_anchor","attest_storage":"https://pith.science/pith/OT6KG5FYO5ANRD3LHWAOSZ4SJR/action/storage_attestation","attest_author":"https://pith.science/pith/OT6KG5FYO5ANRD3LHWAOSZ4SJR/action/author_attestation","sign_citation":"https://pith.science/pith/OT6KG5FYO5ANRD3LHWAOSZ4SJR/action/citation_signature","submit_replication":"https://pith.science/pith/OT6KG5FYO5ANRD3LHWAOSZ4SJR/action/replication_record"}},"created_at":"2026-07-05T11:23:50.702528+00:00","updated_at":"2026-07-05T11:23:50.702528+00:00"}