{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:IEU6O52PZKZVU2QI7BFHEITW5E","short_pith_number":"pith:IEU6O52P","schema_version":"1.0","canonical_sha256":"4129e7774fcab35a6a08f84a722276e93081aeb196f8308a6e4ee4af92093b1d","source":{"kind":"arxiv","id":"2410.18038","version":2},"attestation_state":"computed","paper":{"title":"POD-Attention: Unlocking Full Prefill-Decode Overlap for Faster LLM Inference","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.DC"],"primary_cat":"cs.LG","authors_text":"Aditya K Kamath, Ashish Panwar, Jayashree Mohan, Ramachandran Ramjee, Ramya Prabhu, Simon Peter","submitted_at":"2024-10-23T17:06:56Z","abstract_excerpt":"Each request in LLM inference goes through two phases: compute-bound prefill and memory-bandwidth-bound decode. To improve GPU utilization, recent systems use hybrid batching that combines the prefill and decode phases of different requests into the same batch. This approach optimizes linear operations but remains inefficient for attention computation because existing attention kernels specialize execution independently for the prefill and decode phases.\n  In this paper, we present POD-Attention - the first GPU kernel that efficiently computes attention for hybrid batches. POD-Attention aims t"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.18038","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-10-23T17:06:56Z","cross_cats_sorted":["cs.DC"],"title_canon_sha256":"c0140f9e31874333c34ba36536cc611f08cab4ad19540033c2e842154b5bf7f3","abstract_canon_sha256":"df33e4155cdf51791bfbaf91aad18920a61c0c77dedb4264e9569945ec195213"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:14:47.039741Z","signature_b64":"XLRc3W2vNwmzSZO/suj2JiRvwV/nEhSTp9xU0arO8OQQGGo5ekKFku2aZqBl2QYI1iEouTXv5ozinMnR4PXHCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"4129e7774fcab35a6a08f84a722276e93081aeb196f8308a6e4ee4af92093b1d","last_reissued_at":"2026-07-05T10:14:47.039021Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:14:47.039021Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"POD-Attention: Unlocking Full Prefill-Decode Overlap for Faster LLM Inference","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.DC"],"primary_cat":"cs.LG","authors_text":"Aditya K Kamath, Ashish Panwar, Jayashree Mohan, Ramachandran Ramjee, Ramya Prabhu, Simon Peter","submitted_at":"2024-10-23T17:06:56Z","abstract_excerpt":"Each request in LLM inference goes through two phases: compute-bound prefill and memory-bandwidth-bound decode. To improve GPU utilization, recent systems use hybrid batching that combines the prefill and decode phases of different requests into the same batch. This approach optimizes linear operations but remains inefficient for attention computation because existing attention kernels specialize execution independently for the prefill and decode phases.\n  In this paper, we present POD-Attention - the first GPU kernel that efficiently computes attention for hybrid batches. POD-Attention aims t"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.18038","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.18038/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.18038","created_at":"2026-07-05T10:14:47.039091+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.18038v2","created_at":"2026-07-05T10:14:47.039091+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.18038","created_at":"2026-07-05T10:14:47.039091+00:00"},{"alias_kind":"pith_short_12","alias_value":"IEU6O52PZKZV","created_at":"2026-07-05T10:14:47.039091+00:00"},{"alias_kind":"pith_short_16","alias_value":"IEU6O52PZKZVU2QI","created_at":"2026-07-05T10:14:47.039091+00:00"},{"alias_kind":"pith_short_8","alias_value":"IEU6O52P","created_at":"2026-07-05T10:14:47.039091+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2412.18106","citing_title":"Tackling the Dynamicity in a Production LLM Serving System with SOTA Optimizations via Hybrid Prefill/Decode/Verify Scheduling on Efficient Meta-kernels","ref_index":29,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/IEU6O52PZKZVU2QI7BFHEITW5E","json":"https://pith.science/pith/IEU6O52PZKZVU2QI7BFHEITW5E.json","graph_json":"https://pith.science/api/pith-number/IEU6O52PZKZVU2QI7BFHEITW5E/graph.json","events_json":"https://pith.science/api/pith-number/IEU6O52PZKZVU2QI7BFHEITW5E/events.json","paper":"https://pith.science/paper/IEU6O52P"},"agent_actions":{"view_html":"https://pith.science/pith/IEU6O52PZKZVU2QI7BFHEITW5E","download_json":"https://pith.science/pith/IEU6O52PZKZVU2QI7BFHEITW5E.json","view_paper":"https://pith.science/paper/IEU6O52P","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.18038&json=true","fetch_graph":"https://pith.science/api/pith-number/IEU6O52PZKZVU2QI7BFHEITW5E/graph.json","fetch_events":"https://pith.science/api/pith-number/IEU6O52PZKZVU2QI7BFHEITW5E/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/IEU6O52PZKZVU2QI7BFHEITW5E/action/timestamp_anchor","attest_storage":"https://pith.science/pith/IEU6O52PZKZVU2QI7BFHEITW5E/action/storage_attestation","attest_author":"https://pith.science/pith/IEU6O52PZKZVU2QI7BFHEITW5E/action/author_attestation","sign_citation":"https://pith.science/pith/IEU6O52PZKZVU2QI7BFHEITW5E/action/citation_signature","submit_replication":"https://pith.science/pith/IEU6O52PZKZVU2QI7BFHEITW5E/action/replication_record"}},"created_at":"2026-07-05T10:14:47.039091+00:00","updated_at":"2026-07-05T10:14:47.039091+00:00"}