{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:VFLGLZUBAH6FVW52XVTWBKXXLE","short_pith_number":"pith:VFLGLZUB","schema_version":"1.0","canonical_sha256":"a95665e68101fc5adbbabd6760aaf75907ca1e82b244c85471042efe9b52ce1b","source":{"kind":"arxiv","id":"2503.24358","version":2},"attestation_state":"computed","paper":{"title":"SQuat: Subspace-orthogonal KV Cache Quantization","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.IT","math.IT"],"primary_cat":"cs.LG","authors_text":"Akash Srivastava, Hao Wang, Kai Xu, Ligong Han","submitted_at":"2025-03-31T17:37:32Z","abstract_excerpt":"The key-value (KV) cache accelerates LLMs decoding by storing KV tensors from previously generated tokens. It reduces redundant computation at the cost of increased memory usage. To mitigate this overhead, existing approaches compress KV tensors into lower-bit representations; however, quantization errors can accumulate as more tokens are generated, potentially resulting in undesired outputs. In this paper, we introduce SQuat (Subspace-orthogonal KV cache quantization). It first constructs a subspace spanned by query tensors to capture the most critical task-related information. During key ten"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2503.24358","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.LG","submitted_at":"2025-03-31T17:37:32Z","cross_cats_sorted":["cs.AI","cs.CL","cs.IT","math.IT"],"title_canon_sha256":"f3ef735f8cb33dda6bddce8b8f6e728837c21e8b50030051df4d70a29d7a7c90","abstract_canon_sha256":"2fe43a61e2d37d24da7b63af56326a8a2c2c8011c8a9ca5a9adbebda6a3b11fe"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:44:47.275322Z","signature_b64":"tdogEb8NWZtG3LNUZhQxHYb0Mvhl4bbD+dsVambVw339+va/Uhc8cXqp4D2pRRpDhQZSab0CTXhw8Xsh+WCjAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"a95665e68101fc5adbbabd6760aaf75907ca1e82b244c85471042efe9b52ce1b","last_reissued_at":"2026-07-05T11:44:47.274820Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:44:47.274820Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"SQuat: Subspace-orthogonal KV Cache Quantization","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.IT","math.IT"],"primary_cat":"cs.LG","authors_text":"Akash Srivastava, Hao Wang, Kai Xu, Ligong Han","submitted_at":"2025-03-31T17:37:32Z","abstract_excerpt":"The key-value (KV) cache accelerates LLMs decoding by storing KV tensors from previously generated tokens. It reduces redundant computation at the cost of increased memory usage. To mitigate this overhead, existing approaches compress KV tensors into lower-bit representations; however, quantization errors can accumulate as more tokens are generated, potentially resulting in undesired outputs. In this paper, we introduce SQuat (Subspace-orthogonal KV cache quantization). It first constructs a subspace spanned by query tensors to capture the most critical task-related information. During key ten"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2503.24358","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2503.24358/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2503.24358","created_at":"2026-07-05T11:44:47.274892+00:00"},{"alias_kind":"arxiv_version","alias_value":"2503.24358v2","created_at":"2026-07-05T11:44:47.274892+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2503.24358","created_at":"2026-07-05T11:44:47.274892+00:00"},{"alias_kind":"pith_short_12","alias_value":"VFLGLZUBAH6F","created_at":"2026-07-05T11:44:47.274892+00:00"},{"alias_kind":"pith_short_16","alias_value":"VFLGLZUBAH6FVW52","created_at":"2026-07-05T11:44:47.274892+00:00"},{"alias_kind":"pith_short_8","alias_value":"VFLGLZUB","created_at":"2026-07-05T11:44:47.274892+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.24033","citing_title":"RoPE-Aware Bit Allocation for KV-Cache Quantization","ref_index":36,"is_internal_anchor":false},{"citing_arxiv_id":"2602.23200","citing_title":"InnerQ: Hardware-Aware Tuning-Free Quantization of KV Cache for Large Language Models","ref_index":15,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/VFLGLZUBAH6FVW52XVTWBKXXLE","json":"https://pith.science/pith/VFLGLZUBAH6FVW52XVTWBKXXLE.json","graph_json":"https://pith.science/api/pith-number/VFLGLZUBAH6FVW52XVTWBKXXLE/graph.json","events_json":"https://pith.science/api/pith-number/VFLGLZUBAH6FVW52XVTWBKXXLE/events.json","paper":"https://pith.science/paper/VFLGLZUB"},"agent_actions":{"view_html":"https://pith.science/pith/VFLGLZUBAH6FVW52XVTWBKXXLE","download_json":"https://pith.science/pith/VFLGLZUBAH6FVW52XVTWBKXXLE.json","view_paper":"https://pith.science/paper/VFLGLZUB","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2503.24358&json=true","fetch_graph":"https://pith.science/api/pith-number/VFLGLZUBAH6FVW52XVTWBKXXLE/graph.json","fetch_events":"https://pith.science/api/pith-number/VFLGLZUBAH6FVW52XVTWBKXXLE/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/VFLGLZUBAH6FVW52XVTWBKXXLE/action/timestamp_anchor","attest_storage":"https://pith.science/pith/VFLGLZUBAH6FVW52XVTWBKXXLE/action/storage_attestation","attest_author":"https://pith.science/pith/VFLGLZUBAH6FVW52XVTWBKXXLE/action/author_attestation","sign_citation":"https://pith.science/pith/VFLGLZUBAH6FVW52XVTWBKXXLE/action/citation_signature","submit_replication":"https://pith.science/pith/VFLGLZUBAH6FVW52XVTWBKXXLE/action/replication_record"}},"created_at":"2026-07-05T11:44:47.274892+00:00","updated_at":"2026-07-05T11:44:47.274892+00:00"}