{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:YDKI5NWYH42AEUKSMF5JHTLGRR","short_pith_number":"pith:YDKI5NWY","schema_version":"1.0","canonical_sha256":"c0d48eb6d83f34025152617a93cd668c6f9408d5a7b8dd0e7d7a367eeef10cf0","source":{"kind":"arxiv","id":"2410.03960","version":3},"attestation_state":"computed","paper":{"title":"SwiftKV: Fast Prefill-Optimized Inference with Knowledge-Preserving Model Transformation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Aurick Qiao, Samyam Rajbhandari, Yuxiong He, Zhewei Yao","submitted_at":"2024-10-04T22:45:26Z","abstract_excerpt":"LLM inference for enterprise applications, such as summarization, RAG, and code-generation, typically observe much longer prompt than generations, leading to high prefill cost and response latency. We present SwiftKV, a novel model transformation and distillation procedure targeted at reducing the prefill compute (in FLOPs) of prompt tokens while preserving high generation quality. First, SwiftKV prefills later layers' KV cache using an earlier layer's output, allowing prompt tokens to skip those later layers. Second, SwiftKV employs a lightweight knowledge-preserving distillation procedure th"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.03960","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-10-04T22:45:26Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"47d429d7f94d1e557f7f1e164802aafbfeb3f23b7048c5a85d5889675045b367","abstract_canon_sha256":"b7af2443e5f140a9fa408599887bef1e3678039124d1594a60f6d1b96bb5297c"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:13:58.483438Z","signature_b64":"3jT9bcTGpxN2/g9fMt+JXpmlHjwxfPxJYrYJiM+QsL5yWy1kqzwMq/fPk0otzZ6pVRR/LCCrgBEB5NNsclY5Cw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"c0d48eb6d83f34025152617a93cd668c6f9408d5a7b8dd0e7d7a367eeef10cf0","last_reissued_at":"2026-07-05T11:13:58.482884Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:13:58.482884Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"SwiftKV: Fast Prefill-Optimized Inference with Knowledge-Preserving Model Transformation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Aurick Qiao, Samyam Rajbhandari, Yuxiong He, Zhewei Yao","submitted_at":"2024-10-04T22:45:26Z","abstract_excerpt":"LLM inference for enterprise applications, such as summarization, RAG, and code-generation, typically observe much longer prompt than generations, leading to high prefill cost and response latency. We present SwiftKV, a novel model transformation and distillation procedure targeted at reducing the prefill compute (in FLOPs) of prompt tokens while preserving high generation quality. First, SwiftKV prefills later layers' KV cache using an earlier layer's output, allowing prompt tokens to skip those later layers. Second, SwiftKV employs a lightweight knowledge-preserving distillation procedure th"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.03960","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.03960/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.03960","created_at":"2026-07-05T11:13:58.482942+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.03960v3","created_at":"2026-07-05T11:13:58.482942+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.03960","created_at":"2026-07-05T11:13:58.482942+00:00"},{"alias_kind":"pith_short_12","alias_value":"YDKI5NWYH42A","created_at":"2026-07-05T11:13:58.482942+00:00"},{"alias_kind":"pith_short_16","alias_value":"YDKI5NWYH42AEUKS","created_at":"2026-07-05T11:13:58.482942+00:00"},{"alias_kind":"pith_short_8","alias_value":"YDKI5NWY","created_at":"2026-07-05T11:13:58.482942+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2507.11830","citing_title":"Arctic Inference with Shift Parallelism: Fast and Efficient Open Source Inference System for Enterprise AI","ref_index":14,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/YDKI5NWYH42AEUKSMF5JHTLGRR","json":"https://pith.science/pith/YDKI5NWYH42AEUKSMF5JHTLGRR.json","graph_json":"https://pith.science/api/pith-number/YDKI5NWYH42AEUKSMF5JHTLGRR/graph.json","events_json":"https://pith.science/api/pith-number/YDKI5NWYH42AEUKSMF5JHTLGRR/events.json","paper":"https://pith.science/paper/YDKI5NWY"},"agent_actions":{"view_html":"https://pith.science/pith/YDKI5NWYH42AEUKSMF5JHTLGRR","download_json":"https://pith.science/pith/YDKI5NWYH42AEUKSMF5JHTLGRR.json","view_paper":"https://pith.science/paper/YDKI5NWY","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.03960&json=true","fetch_graph":"https://pith.science/api/pith-number/YDKI5NWYH42AEUKSMF5JHTLGRR/graph.json","fetch_events":"https://pith.science/api/pith-number/YDKI5NWYH42AEUKSMF5JHTLGRR/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/YDKI5NWYH42AEUKSMF5JHTLGRR/action/timestamp_anchor","attest_storage":"https://pith.science/pith/YDKI5NWYH42AEUKSMF5JHTLGRR/action/storage_attestation","attest_author":"https://pith.science/pith/YDKI5NWYH42AEUKSMF5JHTLGRR/action/author_attestation","sign_citation":"https://pith.science/pith/YDKI5NWYH42AEUKSMF5JHTLGRR/action/citation_signature","submit_replication":"https://pith.science/pith/YDKI5NWYH42AEUKSMF5JHTLGRR/action/replication_record"}},"created_at":"2026-07-05T11:13:58.482942+00:00","updated_at":"2026-07-05T11:13:58.482942+00:00"}