{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:F7LOUTKTG5QO3LSAS7SZ7N5BPL","short_pith_number":"pith:F7LOUTKT","schema_version":"1.0","canonical_sha256":"2fd6ea4d533760edae4097e59fb7a17accd6771823e593938c74e778c3cd2cdf","source":{"kind":"arxiv","id":"2502.20766","version":1},"attestation_state":"computed","paper":{"title":"FlexPrefill: A Context-Aware Sparse Attention Mechanism for Efficient Long-Sequence Inference","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.LG","authors_text":"Jianqiao Lu, Xunhao Lai, Xun Zhou, Yao Luo, Yiyuan Ma","submitted_at":"2025-02-28T06:34:53Z","abstract_excerpt":"Large language models (LLMs) encounter computational challenges during long-sequence inference, especially in the attention pre-filling phase, where the complexity grows quadratically with the prompt length. Previous efforts to mitigate these challenges have relied on fixed sparse attention patterns or identifying sparse attention patterns based on limited cases. However, these methods lacked the flexibility to efficiently adapt to varying input demands. In this paper, we introduce FlexPrefill, a Flexible sparse Pre-filling mechanism that dynamically adjusts sparse attention patterns and compu"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.20766","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-02-28T06:34:53Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"39135ab6ef300c205e27be6bc8ab1f78b7dbad2d2ea266b96c7b7b028603ab3d","abstract_canon_sha256":"60ebc3296902be26fc15473006ce2f28ed52156f45f7759241bb5611f518be91"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:21:39.322445Z","signature_b64":"MkNZFL+jGB3wLQuaXnpmPWoFLoGbt5xPqTiCpVWTzLpnozQyu8Qh7JSoV+GKnFjQjTbNxIvkXgUHhmwWrruDAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"2fd6ea4d533760edae4097e59fb7a17accd6771823e593938c74e778c3cd2cdf","last_reissued_at":"2026-07-05T10:21:39.321944Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:21:39.321944Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"FlexPrefill: A Context-Aware Sparse Attention Mechanism for Efficient Long-Sequence Inference","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.LG","authors_text":"Jianqiao Lu, Xunhao Lai, Xun Zhou, Yao Luo, Yiyuan Ma","submitted_at":"2025-02-28T06:34:53Z","abstract_excerpt":"Large language models (LLMs) encounter computational challenges during long-sequence inference, especially in the attention pre-filling phase, where the complexity grows quadratically with the prompt length. Previous efforts to mitigate these challenges have relied on fixed sparse attention patterns or identifying sparse attention patterns based on limited cases. However, these methods lacked the flexibility to efficiently adapt to varying input demands. In this paper, we introduce FlexPrefill, a Flexible sparse Pre-filling mechanism that dynamically adjusts sparse attention patterns and compu"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.20766","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.20766/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.20766","created_at":"2026-07-05T10:21:39.322002+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.20766v1","created_at":"2026-07-05T10:21:39.322002+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.20766","created_at":"2026-07-05T10:21:39.322002+00:00"},{"alias_kind":"pith_short_12","alias_value":"F7LOUTKTG5QO","created_at":"2026-07-05T10:21:39.322002+00:00"},{"alias_kind":"pith_short_16","alias_value":"F7LOUTKTG5QO3LSA","created_at":"2026-07-05T10:21:39.322002+00:00"},{"alias_kind":"pith_short_8","alias_value":"F7LOUTKT","created_at":"2026-07-05T10:21:39.322002+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":17,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.10537","citing_title":"Prefilling-dLLM: Predictive Prefilling for Long-Context Inference in Diffusion Language Models","ref_index":42,"is_internal_anchor":false},{"citing_arxiv_id":"2606.09441","citing_title":"SIFT: Selective-Index For Fast Compute of RAG Prefill by Exploiting Attention Invariance","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2606.28379","citing_title":"LEDGER: Scaling Agentic Document Editing with Dependency-aware Graph Retrieval","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2605.23445","citing_title":"DFSAttn: Dynamic Fine-grained Sparse Attention for Efficient Video Generation","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2602.18196","citing_title":"RAT+: Train Dense, Infer Sparse -- Recurrence Augmented Attention for Dilated Inference","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2605.19893","citing_title":"SSV: Sparse Speculative Verification for Efficient LLM Inference","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2605.20813","citing_title":"PulseCol: Periodically Refreshed Column-Sparse Attention for Accelerating Diffusion Language Models","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18071","citing_title":"KVDrive: A Holistic Multi-Tier KV Cache Management System for Long-Context LLM Inference","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2605.16839","citing_title":"CompactAttention: Accelerating Chunked Prefill with Block-Union KV Selection","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2507.21526","citing_title":"Accelerating Prefilling via Decoding-time Contribution Sparsity","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2508.16703","citing_title":"ShadowNPU: System and Algorithm Co-design for NPU-Centric On-Device LLM Inference","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2512.12087","citing_title":"BLASST: Dynamic BLocked Attention Sparsity via Softmax Thresholding","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2602.18196","citing_title":"RAT+: Train Dense, Infer Sparse -- Recurrence Augmented Attention for Dilated Inference","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08840","citing_title":"ReST-KV: Robust KV Cache Eviction with Layer-wise Output Reconstruction and Spatial-Temporal Smoothing","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2604.24820","citing_title":"Salca: A Sparsity-Aware Hardware Accelerator for Efficient Long-Context Attention Decoding","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07719","citing_title":"An Efficient Hybrid Sparse Attention with CPU-GPU Parallelism for Long-Context Inference","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06763","citing_title":"Sparse Attention as a Range Searching Problem: Towards an Inference-Efficient Index for KV Cache","ref_index":28,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/F7LOUTKTG5QO3LSAS7SZ7N5BPL","json":"https://pith.science/pith/F7LOUTKTG5QO3LSAS7SZ7N5BPL.json","graph_json":"https://pith.science/api/pith-number/F7LOUTKTG5QO3LSAS7SZ7N5BPL/graph.json","events_json":"https://pith.science/api/pith-number/F7LOUTKTG5QO3LSAS7SZ7N5BPL/events.json","paper":"https://pith.science/paper/F7LOUTKT"},"agent_actions":{"view_html":"https://pith.science/pith/F7LOUTKTG5QO3LSAS7SZ7N5BPL","download_json":"https://pith.science/pith/F7LOUTKTG5QO3LSAS7SZ7N5BPL.json","view_paper":"https://pith.science/paper/F7LOUTKT","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.20766&json=true","fetch_graph":"https://pith.science/api/pith-number/F7LOUTKTG5QO3LSAS7SZ7N5BPL/graph.json","fetch_events":"https://pith.science/api/pith-number/F7LOUTKTG5QO3LSAS7SZ7N5BPL/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/F7LOUTKTG5QO3LSAS7SZ7N5BPL/action/timestamp_anchor","attest_storage":"https://pith.science/pith/F7LOUTKTG5QO3LSAS7SZ7N5BPL/action/storage_attestation","attest_author":"https://pith.science/pith/F7LOUTKTG5QO3LSAS7SZ7N5BPL/action/author_attestation","sign_citation":"https://pith.science/pith/F7LOUTKTG5QO3LSAS7SZ7N5BPL/action/citation_signature","submit_replication":"https://pith.science/pith/F7LOUTKTG5QO3LSAS7SZ7N5BPL/action/replication_record"}},"created_at":"2026-07-05T10:21:39.322002+00:00","updated_at":"2026-07-05T10:21:39.322002+00:00"}