{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:HTTTMZ6PCHE2EUSKEVTVHDSCU7","short_pith_number":"pith:HTTTMZ6P","schema_version":"1.0","canonical_sha256":"3ce73667cf11c9a2524a2567538e42a7c11274e6be66ee7cc7c5ad51ce2c6c05","source":{"kind":"arxiv","id":"2411.13820","version":2},"attestation_state":"computed","paper":{"title":"InstCache: A Predictive Cache for LLM Serving","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.DC"],"primary_cat":"cs.CL","authors_text":"Jiamu Kang, Jiangang Kong, Longwei Zou, Tingfeng Liu, Yangdong Deng, Yan Liu","submitted_at":"2024-11-21T03:52:41Z","abstract_excerpt":"The revolutionary capabilities of Large Language Models (LLMs) are attracting rapidly growing popularity and leading to soaring user requests to inference serving systems. Caching techniques, which leverage data reuse to reduce computation, offer opportunities to optimize the performance of LLM inference engines. On the one hand, the low-level key-value (KV) cache working at the token level is widely adopted, albeit it incurs significant overhead as request volume grows. On the other hand, instruction-level caching, which stores full instruction-response pairs, is expected to play an increasin"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2411.13820","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-11-21T03:52:41Z","cross_cats_sorted":["cs.DC"],"title_canon_sha256":"12848c07fd2500f05b37ba3a744050f51180a763ab4dfe9a886c52fbc4500772","abstract_canon_sha256":"136c9493182a0719499a5f0b84e0cb803fcff4fee97ee6a61031c10027067c76"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:36:25.728270Z","signature_b64":"puAdN5+TFwl2pLHD1COo2N00AQl3T/sl5o2NL+qh/aGUf07CIflUjBaZfmDrdBua6Pi2WUi91O0ydBvHbxxwDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"3ce73667cf11c9a2524a2567538e42a7c11274e6be66ee7cc7c5ad51ce2c6c05","last_reissued_at":"2026-07-05T11:36:25.727768Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:36:25.727768Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"InstCache: A Predictive Cache for LLM Serving","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.DC"],"primary_cat":"cs.CL","authors_text":"Jiamu Kang, Jiangang Kong, Longwei Zou, Tingfeng Liu, Yangdong Deng, Yan Liu","submitted_at":"2024-11-21T03:52:41Z","abstract_excerpt":"The revolutionary capabilities of Large Language Models (LLMs) are attracting rapidly growing popularity and leading to soaring user requests to inference serving systems. Caching techniques, which leverage data reuse to reduce computation, offer opportunities to optimize the performance of LLM inference engines. On the one hand, the low-level key-value (KV) cache working at the token level is widely adopted, albeit it incurs significant overhead as request volume grows. On the other hand, instruction-level caching, which stores full instruction-response pairs, is expected to play an increasin"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2411.13820","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2411.13820/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2411.13820","created_at":"2026-07-05T11:36:25.727832+00:00"},{"alias_kind":"arxiv_version","alias_value":"2411.13820v2","created_at":"2026-07-05T11:36:25.727832+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2411.13820","created_at":"2026-07-05T11:36:25.727832+00:00"},{"alias_kind":"pith_short_12","alias_value":"HTTTMZ6PCHE2","created_at":"2026-07-05T11:36:25.727832+00:00"},{"alias_kind":"pith_short_16","alias_value":"HTTTMZ6PCHE2EUSK","created_at":"2026-07-05T11:36:25.727832+00:00"},{"alias_kind":"pith_short_8","alias_value":"HTTTMZ6P","created_at":"2026-07-05T11:36:25.727832+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/HTTTMZ6PCHE2EUSKEVTVHDSCU7","json":"https://pith.science/pith/HTTTMZ6PCHE2EUSKEVTVHDSCU7.json","graph_json":"https://pith.science/api/pith-number/HTTTMZ6PCHE2EUSKEVTVHDSCU7/graph.json","events_json":"https://pith.science/api/pith-number/HTTTMZ6PCHE2EUSKEVTVHDSCU7/events.json","paper":"https://pith.science/paper/HTTTMZ6P"},"agent_actions":{"view_html":"https://pith.science/pith/HTTTMZ6PCHE2EUSKEVTVHDSCU7","download_json":"https://pith.science/pith/HTTTMZ6PCHE2EUSKEVTVHDSCU7.json","view_paper":"https://pith.science/paper/HTTTMZ6P","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2411.13820&json=true","fetch_graph":"https://pith.science/api/pith-number/HTTTMZ6PCHE2EUSKEVTVHDSCU7/graph.json","fetch_events":"https://pith.science/api/pith-number/HTTTMZ6PCHE2EUSKEVTVHDSCU7/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/HTTTMZ6PCHE2EUSKEVTVHDSCU7/action/timestamp_anchor","attest_storage":"https://pith.science/pith/HTTTMZ6PCHE2EUSKEVTVHDSCU7/action/storage_attestation","attest_author":"https://pith.science/pith/HTTTMZ6PCHE2EUSKEVTVHDSCU7/action/author_attestation","sign_citation":"https://pith.science/pith/HTTTMZ6PCHE2EUSKEVTVHDSCU7/action/citation_signature","submit_replication":"https://pith.science/pith/HTTTMZ6PCHE2EUSKEVTVHDSCU7/action/replication_record"}},"created_at":"2026-07-05T11:36:25.727832+00:00","updated_at":"2026-07-05T11:36:25.727832+00:00"}