{"as_of":"2026-08-16T07:36:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:3d4f311832eef233562f18d49605af3b0974eee296bb8bb2a79c6d5205e80a7b","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":37,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":37,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-16T06:30:59.297886+00:00","state":"measured"},{"denominator":37,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":37,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-15T18:40:37.603820Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"pith","source_observed_at":"2026-07-10T01:36:44.121083Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2407.21118","last_updated":"2024-11-04T02:08:55Z","snapshot_observed_at":"2026-08-12T23:11:46.275639Z","submitted_at":"2024-07-30T18:19:38Z","title":"Palu: Compressing KV-Cache with Low-Rank Projection","version":2},"cited_work":{"arxiv_id":"2407.21118","doi":null,"metadata_source":"pith","pith_arxiv_id":"2407.21118","snapshot_observed_at":"2026-07-10T01:36:44.121083Z","title":"Palu: Compressing kv-cache with low-rank projection.arXiv preprint arXiv:2407.21118","venue":"cs.AI","work_id":"dc6247ce-a2da-431d-a935-3fb13543cb13","year":2024},"citing_paper":{"arxiv_id":"2312.05821","last_updated":"2025-08-28T03:57:52Z","snapshot_observed_at":"2026-08-15T15:15:36.775697Z","submitted_at":"2023-12-10T08:41:24Z","title":"ASVD: Activation-aware Singular Value Decomposition for Compressing Large Language Models","version":5},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-20T13:49:33.747672Z"},"links":{"cited_paper":"/paper/2407.21118","citing_paper":"/paper/2312.05821"},"observation_digest":"sha256:0a8d24353275218a46be711950d9caaf293725a34152be68dc744951647859ce","observation_id":"f73c71fa-3690-4967-8845-19cad3674fec","resolution":{"observed_at":"2026-05-20T13:49:33.802941Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.21118","last_updated":"2024-11-04T02:08:55Z","snapshot_observed_at":"2026-08-12T23:11:46.275639Z","submitted_at":"2024-07-30T18:19:38Z","title":"Palu: Compressing KV-Cache with Low-Rank Projection","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.21118","snapshot_observed_at":"2026-08-12T11:56:33.034655Z","title":"S., and Wu, K.-C","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2411.17685","last_updated":"2024-11-26T18:52:06Z","snapshot_observed_at":"2026-08-13T12:25:42.413667Z","submitted_at":"2024-11-26T18:52:06Z","title":"Attamba: Attending To Multi-Token States","version":1},"reference_index":2,"source":"arxiv_source","source_observed_at":"2026-08-12T11:56:33.034655Z"},"links":{"cited_paper":"/paper/2407.21118","citing_paper":"/paper/2411.17685"},"observation_digest":"sha256:e7f5b79fd21138c64839f76aa2ca4f22201f5e16a39924a66725b84adab0bd3b","observation_id":"e3967fdd-1b7c-46d2-ac7e-5fb0b7207041","resolution":{"observed_at":"2026-08-12T11:56:33.034655Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.21118","last_updated":"2024-11-04T02:08:55Z","snapshot_observed_at":"2026-08-12T23:11:46.275639Z","submitted_at":"2024-07-30T18:19:38Z","title":"Palu: Compressing KV-Cache with Low-Rank Projection","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.21118","snapshot_observed_at":"2026-08-11T12:22:38.776013Z","title":"S., and Wu, K.-C","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.14363","last_updated":"2025-02-03T21:45:32Z","snapshot_observed_at":"2026-08-13T00:10:37.377984Z","submitted_at":"2024-12-18T22:01:55Z","title":"ResQ: Mixed-Precision Quantization of Large Language Models with Low-Rank Residuals","version":2},"reference_index":6,"source":"arxiv_source","source_observed_at":"2026-08-11T12:22:38.776013Z"},"links":{"cited_paper":"/paper/2407.21118","citing_paper":"/paper/2412.14363"},"observation_digest":"sha256:69d39acb5fe85fe1c7255d41290e3e60add86027a2b5070ef4bcdcfe42aa4908","observation_id":"323c0ab9-4f1b-4b74-ad4e-9bb33b0d543a","resolution":{"observed_at":"2026-08-11T12:22:38.776013Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.21118","last_updated":"2024-11-04T02:08:55Z","snapshot_observed_at":"2026-08-12T23:11:46.275639Z","submitted_at":"2024-07-30T18:19:38Z","title":"Palu: Compressing KV-Cache with Low-Rank Projection","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.21118","snapshot_observed_at":"2026-08-11T00:38:46.985213Z","title":"Palu: Compressing kv-cache with low-rank projection,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.19442","last_updated":"2025-07-30T05:24:46Z","snapshot_observed_at":"2026-08-15T13:24:51.697670Z","submitted_at":"2024-12-27T04:17:57Z","title":"A Survey on Large Language Model Acceleration based on KV Cache Management","version":3},"reference_index":72,"source":"pdf_text","source_observed_at":"2026-08-11T00:38:46.985213Z"},"links":{"cited_paper":"/paper/2407.21118","citing_paper":"/paper/2412.19442"},"observation_digest":"sha256:76baf55ad858bae008eb1ba4aaaced7518a258f269d686a87e32b1546dcb2590","observation_id":"3ff7df90-6c6e-49d1-80ec-2aa6f8355212","resolution":{"observed_at":"2026-08-11T00:38:46.985213Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.21118","last_updated":"2024-11-04T02:08:55Z","snapshot_observed_at":"2026-08-12T23:11:46.275639Z","submitted_at":"2024-07-30T18:19:38Z","title":"Palu: Compressing KV-Cache with Low-Rank Projection","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.21118","snapshot_observed_at":"2026-08-09T20:34:01.189891Z","title":"Palu: Compressing kv-cache with low-rank projection","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.19392","last_updated":"2025-02-28T18:04:52Z","snapshot_observed_at":"2026-08-12T17:04:32.510964Z","submitted_at":"2025-01-31T18:47:42Z","title":"Cache Me If You Must: Adaptive Key-Value Quantization for Large Language Models","version":4},"reference_index":7,"source":"arxiv_source","source_observed_at":"2026-08-09T20:34:01.189891Z"},"links":{"cited_paper":"/paper/2407.21118","citing_paper":"/paper/2501.19392"},"observation_digest":"sha256:fdae7edb3dfd0c927e6b8117c905e846fa4a33e33bf06b1d810b71b0a1837d36","observation_id":"9e72fe70-0f02-408e-ad53-19ad07a7a089","resolution":{"observed_at":"2026-08-09T20:34:01.189891Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.21118","last_updated":"2024-11-04T02:08:55Z","snapshot_observed_at":"2026-08-12T23:11:46.275639Z","submitted_at":"2024-07-30T18:19:38Z","title":"Palu: Compressing KV-Cache with Low-Rank Projection","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.21118","snapshot_observed_at":"2026-08-08T11:48:01.171526Z","title":"Palu: Compressing kv-cache with low-rank projection","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.07864","last_updated":"2025-06-12T11:45:57Z","snapshot_observed_at":"2026-08-10T11:58:20.177769Z","submitted_at":"2025-02-11T18:20:18Z","title":"TransMLA: Multi-Head Latent Attention Is All You Need","version":5},"reference_index":22,"source":"arxiv_source","source_observed_at":"2026-08-08T11:48:01.171526Z"},"links":{"cited_paper":"/paper/2407.21118","citing_paper":"/paper/2502.07864"},"observation_digest":"sha256:4193d0edffa5ade1ebf5280658c6354c2302b17ce78792db7ba72ef39460e04d","observation_id":"def58ec3-b6a0-4d23-a0ba-3f4148078342","resolution":{"observed_at":"2026-08-08T11:48:01.171526Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.21118","last_updated":"2024-11-04T02:08:55Z","snapshot_observed_at":"2026-08-12T23:11:46.275639Z","submitted_at":"2024-07-30T18:19:38Z","title":"Palu: Compressing KV-Cache with Low-Rank Projection","version":2},"cited_work":{"arxiv_id":"2407.21118","doi":null,"metadata_source":"pith","pith_arxiv_id":"2407.21118","snapshot_observed_at":"2026-07-10T01:36:44.121083Z","title":"Palu: Compressing kv-cache with low-rank projection.arXiv preprint arXiv:2407.21118","venue":"cs.AI","work_id":"dc6247ce-a2da-431d-a935-3fb13543cb13","year":2024},"citing_paper":{"arxiv_id":"2505.12942","last_updated":"2026-05-12T22:54:57Z","snapshot_observed_at":"2026-08-15T02:39:36.208018Z","submitted_at":"2025-05-19T10:29:32Z","title":"A3 : an Analytical Low-Rank Approximation Framework for Attention","version":4},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-22T15:02:33.864861Z"},"links":{"cited_paper":"/paper/2407.21118","citing_paper":"/paper/2505.12942"},"observation_digest":"sha256:5266b973fca0c416afde46dad21e04448d839f0af6ab4f34bf09fc4180160d4c","observation_id":"f857c088-56f1-40a0-b779-0614274d8b18","resolution":{"observed_at":"2026-05-22T15:04:57.131560Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.21118","last_updated":"2024-11-04T02:08:55Z","snapshot_observed_at":"2026-08-12T23:11:46.275639Z","submitted_at":"2024-07-30T18:19:38Z","title":"Palu: Compressing KV-Cache with Low-Rank Projection","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.21118","snapshot_observed_at":"2026-08-07T14:36:25.552150Z","title":"Palu: Compressing KV-cache with low-rank projection","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.18413","last_updated":"2025-05-23T22:39:54Z","snapshot_observed_at":"2026-08-16T06:44:53.557703Z","submitted_at":"2025-05-23T22:39:54Z","title":"LatentLLM: Attention-Aware Joint Tensor Compression","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-07T14:36:25.552150Z"},"links":{"cited_paper":"/paper/2407.21118","citing_paper":"/paper/2505.18413"},"observation_digest":"sha256:c5721411dc362b00971507f8087b779a95417ad3b6f53d35d65fd3f894de2733","observation_id":"de377a09-f03b-457f-b1bd-0eb0dffa5fe3","resolution":{"observed_at":"2026-08-07T14:36:25.552150Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.21118","last_updated":"2024-11-04T02:08:55Z","snapshot_observed_at":"2026-08-12T23:11:46.275639Z","submitted_at":"2024-07-30T18:19:38Z","title":"Palu: Compressing KV-Cache with Low-Rank Projection","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.21118","snapshot_observed_at":"2026-08-07T13:32:30.031025Z","title":"Abdelfattah, and Kai-Chiang Wu","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.21487","last_updated":"2025-05-27T17:54:07Z","snapshot_observed_at":"2026-08-07T13:24:37.243429Z","submitted_at":"2025-05-27T17:54:07Z","title":"Hardware-Efficient Attention for Fast Decoding","version":1},"reference_index":12,"source":"arxiv_source","source_observed_at":"2026-08-07T13:32:30.031025Z"},"links":{"cited_paper":"/paper/2407.21118","citing_paper":"/paper/2505.21487"},"observation_digest":"sha256:c438770cbae87c129652906fe50c17491adc4be1cb26e183584c7c32f1ab7c3a","observation_id":"7e5ea5fb-a49e-40fb-b73a-2d4c9cb73d6a","resolution":{"observed_at":"2026-08-07T13:32:30.031025Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.21118","last_updated":"2024-11-04T02:08:55Z","snapshot_observed_at":"2026-08-12T23:11:46.275639Z","submitted_at":"2024-07-30T18:19:38Z","title":"Palu: Compressing KV-Cache with Low-Rank Projection","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.21118","snapshot_observed_at":"2026-08-07T10:45:09.825323Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.04642","last_updated":"2025-06-05T05:23:38Z","snapshot_observed_at":"2026-08-11T12:22:25.485792Z","submitted_at":"2025-06-05T05:23:38Z","title":"TaDA: Training-free recipe for Decoding with Adaptive KV Cache Compression and Mean-centering","version":1},"reference_index":3,"source":"arxiv_source","source_observed_at":"2026-08-07T10:45:09.825323Z"},"links":{"cited_paper":"/paper/2407.21118","citing_paper":"/paper/2506.04642"},"observation_digest":"sha256:abe92dcc41ca2793bd1f13f6b3f570961b620b311284146d4db78f85c02ae1b6","observation_id":"484ef146-5291-4782-9dd0-fdc49ba24a4b","resolution":{"observed_at":"2026-08-07T10:45:09.825323Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.21118","last_updated":"2024-11-04T02:08:55Z","snapshot_observed_at":"2026-08-12T23:11:46.275639Z","submitted_at":"2024-07-30T18:19:38Z","title":"Palu: Compressing KV-Cache with Low-Rank Projection","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.21118","snapshot_observed_at":"2026-08-07T06:04:26.941477Z","title":"Palu: Compressing kv-cache with low-rank projection.arXiv preprint arXiv:2407.21118, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.06266","last_updated":"2025-06-13T17:58:55Z","snapshot_observed_at":"2026-08-16T06:23:05.141323Z","submitted_at":"2025-06-06T17:48:23Z","title":"Cartridges: Lightweight and general-purpose long context representations via self-study","version":3},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-07T06:04:26.941477Z"},"links":{"cited_paper":"/paper/2407.21118","citing_paper":"/paper/2506.06266"},"observation_digest":"sha256:6d72d247719c8fbf2442e4d84bd49a1ee10455110d66e19282b15f2bcacd3ef7","observation_id":"6b808613-58d1-4cf3-b466-da47f926ab65","resolution":{"observed_at":"2026-08-07T06:04:26.941477Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.21118","last_updated":"2024-11-04T02:08:55Z","snapshot_observed_at":"2026-08-12T23:11:46.275639Z","submitted_at":"2024-07-30T18:19:38Z","title":"Palu: Compressing KV-Cache with Low-Rank Projection","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.21118","snapshot_observed_at":"2026-08-07T01:10:37.292088Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.11886","last_updated":"2025-06-13T15:35:54Z","snapshot_observed_at":"2026-08-13T15:27:01.467148Z","submitted_at":"2025-06-13T15:35:54Z","title":"Beyond Homogeneous Attention: Memory-Efficient LLMs via Fourier-Approximated KV Cache","version":1},"reference_index":4,"source":"arxiv_source","source_observed_at":"2026-08-07T01:10:37.292088Z"},"links":{"cited_paper":"/paper/2407.21118","citing_paper":"/paper/2506.11886"},"observation_digest":"sha256:df23c976b5d651e36f6e50b76e045023dcb1a58fbc58410b65537f0f2bd98366","observation_id":"77689f63-3157-4247-bc97-f1a0c49a8c57","resolution":{"observed_at":"2026-08-07T01:10:37.292088Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.21118","last_updated":"2024-11-04T02:08:55Z","snapshot_observed_at":"2026-08-12T23:11:46.275639Z","submitted_at":"2024-07-30T18:19:38Z","title":"Palu: Compressing KV-Cache with Low-Rank Projection","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.21118","snapshot_observed_at":"2026-08-15T18:40:37.603820Z","title":"Palu: Compressing kv-cache with low-rank projection.arXiv preprint arXiv:2407.21118, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.19549","last_updated":"2025-06-24T11:55:43Z","snapshot_observed_at":"2026-08-15T18:29:12.724199Z","submitted_at":"2025-06-24T11:55:43Z","title":"RCStat: A Statistical Framework for using Relative Contextualization in Transformers","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-15T18:40:37.603820Z"},"links":{"cited_paper":"/paper/2407.21118","citing_paper":"/paper/2506.19549"},"observation_digest":"sha256:96b00680178fb4b01d5d462bff9bc319ffd93a6df35b45d283693ce8db09636b","observation_id":"11c2edea-ccd7-4683-aea8-76ea40ffa21c","resolution":{"observed_at":"2026-08-15T18:40:37.603820Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.21118","last_updated":"2024-11-04T02:08:55Z","snapshot_observed_at":"2026-08-12T23:11:46.275639Z","submitted_at":"2024-07-30T18:19:38Z","title":"Palu: Compressing KV-Cache with Low-Rank Projection","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.21118","snapshot_observed_at":"2026-08-06T13:57:15.670799Z","title":"Palu: Compressing kv-cache with low-rank projection","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.19906","last_updated":"2025-08-04T08:19:26Z","snapshot_observed_at":"2026-08-12T20:33:33.893453Z","submitted_at":"2025-07-26T10:34:53Z","title":"CaliDrop: KV Cache Compression with Calibration","version":2},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-06T13:57:15.670799Z"},"links":{"cited_paper":"/paper/2407.21118","citing_paper":"/paper/2507.19906"},"observation_digest":"sha256:ea73002f2235119565fa155ab0286bcefc880df6d06fed308dc343a5bacc6a5d","observation_id":"72c9b27c-2c3b-4a5d-87d3-bc691b0d4124","resolution":{"observed_at":"2026-08-06T13:57:15.670799Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.21118","last_updated":"2024-11-04T02:08:55Z","snapshot_observed_at":"2026-08-12T23:11:46.275639Z","submitted_at":"2024-07-30T18:19:38Z","title":"Palu: Compressing KV-Cache with Low-Rank Projection","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.21118","snapshot_observed_at":"2026-08-05T17:55:06.348432Z","title":"Palu: Compressing kv-cache with low-rank projection","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.15881","last_updated":"2025-08-25T02:24:20Z","snapshot_observed_at":"2026-08-05T17:55:04.790166Z","submitted_at":"2025-08-21T15:25:40Z","title":"TPLA: Tensor Parallel Latent Attention for Efficient Disaggregated Prefill and Decode Inference","version":2},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-05T17:55:06.348432Z"},"links":{"cited_paper":"/paper/2407.21118","citing_paper":"/paper/2508.15881"},"observation_digest":"sha256:0e408f52e5eee6bc60f13736514b9c77b983ec166d4a250ee7f4ed7a598b6291","observation_id":"f929d593-9219-4713-9a8b-eef18834e01c","resolution":{"observed_at":"2026-08-05T17:55:06.348432Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.21118","last_updated":"2024-11-04T02:08:55Z","snapshot_observed_at":"2026-08-12T23:11:46.275639Z","submitted_at":"2024-07-30T18:19:38Z","title":"Palu: Compressing KV-Cache with Low-Rank Projection","version":2},"cited_work":{"arxiv_id":"2407.21118","doi":null,"metadata_source":"pith","pith_arxiv_id":"2407.21118","snapshot_observed_at":"2026-07-10T01:36:44.121083Z","title":"Palu: Compressing kv-cache with low-rank projection.arXiv preprint arXiv:2407.21118","venue":"cs.AI","work_id":"dc6247ce-a2da-431d-a935-3fb13543cb13","year":2024},"citing_paper":{"arxiv_id":"2509.21623","last_updated":"2026-04-16T21:29:54Z","snapshot_observed_at":"2026-08-14T12:21:00.182138Z","submitted_at":"2025-09-25T21:42:27Z","title":"OjaKV: Context-Aware Online Low-Rank KV Cache Compression","version":2},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-18T13:26:02.980973Z"},"links":{"cited_paper":"/paper/2407.21118","citing_paper":"/paper/2509.21623"},"observation_digest":"sha256:c6ab778697abf821ca30eb11c0fb4b55186303fdb8021566d1d3bbde68307096","observation_id":"98b382e6-6302-4125-9fce-9adea19682cf","resolution":{"observed_at":"2026-05-18T13:26:24.651969Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.21118","last_updated":"2024-11-04T02:08:55Z","snapshot_observed_at":"2026-08-12T23:11:46.275639Z","submitted_at":"2024-07-30T18:19:38Z","title":"Palu: Compressing KV-Cache with Low-Rank Projection","version":2},"cited_work":{"arxiv_id":"2407.21118","doi":null,"metadata_source":"pith","pith_arxiv_id":"2407.21118","snapshot_observed_at":"2026-07-10T01:36:44.121083Z","title":"Palu: Compressing kv-cache with low-rank projection.arXiv preprint arXiv:2407.21118","venue":"cs.AI","work_id":"dc6247ce-a2da-431d-a935-3fb13543cb13","year":2024},"citing_paper":{"arxiv_id":"2603.22910","last_updated":"2026-05-13T05:25:05Z","snapshot_observed_at":"2026-08-13T08:00:34.413924Z","submitted_at":"2026-03-24T07:58:42Z","title":"EchoKV: Efficient KV Cache Compression via Similarity-Based Reconstruction","version":2},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-15T01:09:07.983785Z"},"links":{"cited_paper":"/paper/2407.21118","citing_paper":"/paper/2603.22910"},"observation_digest":"sha256:469ff84cd479efaa2e1d23ba6a21f6d81975f96fcb3610df252acfb47067196c","observation_id":"57baefaa-2193-4053-b8d4-b8587bdf6c13","resolution":{"observed_at":"2026-05-15T01:09:36.971195Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.21118","last_updated":"2024-11-04T02:08:55Z","snapshot_observed_at":"2026-08-12T23:11:46.275639Z","submitted_at":"2024-07-30T18:19:38Z","title":"Palu: Compressing KV-Cache with Low-Rank Projection","version":2},"cited_work":{"arxiv_id":"2407.21118","doi":null,"metadata_source":"pith","pith_arxiv_id":"2407.21118","snapshot_observed_at":"2026-07-10T01:36:44.121083Z","title":"Palu: Compressing kv-cache with low-rank projection.arXiv preprint arXiv:2407.21118","venue":"cs.AI","work_id":"dc6247ce-a2da-431d-a935-3fb13543cb13","year":2024},"citing_paper":{"arxiv_id":"2604.02570","last_updated":"2026-04-02T22:49:57Z","snapshot_observed_at":"2026-08-12T12:43:33.554905Z","submitted_at":"2026-04-02T22:49:57Z","title":"WSVD: Weighted Low-Rank Approximation for Fast and Efficient Execution of Low-Precision Vision-Language Models","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-13T21:05:09.254086Z"},"links":{"cited_paper":"/paper/2407.21118","citing_paper":"/paper/2604.02570"},"observation_digest":"sha256:a48b7afefdc2346fc5297834b02de00162b79b28907ec7d1a2e3e112650f689c","observation_id":"7ea17c51-bd2f-45bc-9051-11e8fed10732","resolution":{"observed_at":"2026-05-13T21:08:18.020672Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.21118","last_updated":"2024-11-04T02:08:55Z","snapshot_observed_at":"2026-08-12T23:11:46.275639Z","submitted_at":"2024-07-30T18:19:38Z","title":"Palu: Compressing KV-Cache with Low-Rank Projection","version":2},"cited_work":{"arxiv_id":"2407.21118","doi":null,"metadata_source":"pith","pith_arxiv_id":"2407.21118","snapshot_observed_at":"2026-07-10T01:36:44.121083Z","title":"Palu: Compressing kv-cache with low-rank projection.arXiv preprint arXiv:2407.21118","venue":"cs.AI","work_id":"dc6247ce-a2da-431d-a935-3fb13543cb13","year":2024},"citing_paper":{"arxiv_id":"2605.02905","last_updated":"2026-04-06T02:05:52Z","snapshot_observed_at":"2026-08-15T17:43:15.092671Z","submitted_at":"2026-04-06T02:05:52Z","title":"eOptShrinkQ: Near-Lossless KV Cache Compression Through Optimal Spectral Denoising and Quantization","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-05-10T20:18:04.392331Z"},"links":{"cited_paper":"/paper/2407.21118","citing_paper":"/paper/2605.02905"},"observation_digest":"sha256:02b09a9deb36b6fb9eb9b8153c465d0e36c0d5c2516ab8c13d52782ae8a492aa","observation_id":"e152ac02-d2c4-459e-8575-17a5922a051c","resolution":{"observed_at":"2026-05-10T22:00:50.983412Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.21118","last_updated":"2024-11-04T02:08:55Z","snapshot_observed_at":"2026-08-12T23:11:46.275639Z","submitted_at":"2024-07-30T18:19:38Z","title":"Palu: Compressing KV-Cache with Low-Rank Projection","version":2},"cited_work":{"arxiv_id":"2407.21118","doi":null,"metadata_source":"pith","pith_arxiv_id":"2407.21118","snapshot_observed_at":"2026-07-10T01:36:44.121083Z","title":"Palu: Compressing kv-cache with low-rank projection.arXiv preprint arXiv:2407.21118","venue":"cs.AI","work_id":"dc6247ce-a2da-431d-a935-3fb13543cb13","year":2024},"citing_paper":{"arxiv_id":"2605.17757","last_updated":"2026-05-18T02:24:29Z","snapshot_observed_at":"2026-08-15T00:39:44.412521Z","submitted_at":"2026-05-18T02:24:29Z","title":"OSCAR: Offline Spectral Covariance-Aware Rotation for 2-bit KV Cache Quantization","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-05-20T12:16:07.797702Z"},"links":{"cited_paper":"/paper/2407.21118","citing_paper":"/paper/2605.17757"},"observation_digest":"sha256:a4a1c76f5e1edc87a19820aea3d80bde22aa777f7d9ee3dc81ea6bd8f78f0c7b","observation_id":"f5475433-8212-4511-bf65-d1c982f6ba30","resolution":{"observed_at":"2026-05-20T12:18:16.527054Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.21118","last_updated":"2024-11-04T02:08:55Z","snapshot_observed_at":"2026-08-12T23:11:46.275639Z","submitted_at":"2024-07-30T18:19:38Z","title":"Palu: Compressing KV-Cache with Low-Rank Projection","version":2},"cited_work":{"arxiv_id":"2407.21118","doi":null,"metadata_source":"pith","pith_arxiv_id":"2407.21118","snapshot_observed_at":"2026-07-10T01:36:44.121083Z","title":"Palu: Compressing kv-cache with low-rank projection.arXiv preprint arXiv:2407.21118","venue":"cs.AI","work_id":"dc6247ce-a2da-431d-a935-3fb13543cb13","year":2024},"citing_paper":{"arxiv_id":"2605.19218","last_updated":"2026-05-19T00:45:00Z","snapshot_observed_at":"2026-08-13T04:53:15.094121Z","submitted_at":"2026-05-19T00:45:00Z","title":"Rotation-Aligned Key Channel Pruning for Efficient Vision-Language Model Inference","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-05-20T07:43:18.828740Z"},"links":{"cited_paper":"/paper/2407.21118","citing_paper":"/paper/2605.19218"},"observation_digest":"sha256:0c69ff8ee9b5ed5b3f81ccc509a7de833d743d39a05685b5b17fa6bccf1fd9f9","observation_id":"9ab8b0ee-4d6c-4b4b-b1ec-b4cf79efc620","resolution":{"observed_at":"2026-05-20T07:43:23.770713Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.21118","last_updated":"2024-11-04T02:08:55Z","snapshot_observed_at":"2026-08-12T23:11:46.275639Z","submitted_at":"2024-07-30T18:19:38Z","title":"Palu: Compressing KV-Cache with Low-Rank Projection","version":2},"cited_work":{"arxiv_id":"2407.21118","doi":null,"metadata_source":"pith","pith_arxiv_id":"2407.21118","snapshot_observed_at":"2026-07-10T01:36:44.121083Z","title":"Palu: Compressing kv-cache with low-rank projection.arXiv preprint arXiv:2407.21118","venue":"cs.AI","work_id":"dc6247ce-a2da-431d-a935-3fb13543cb13","year":2024},"citing_paper":{"arxiv_id":"2605.31105","last_updated":"2026-05-29T10:16:30Z","snapshot_observed_at":"2026-08-15T17:36:31.606537Z","submitted_at":"2026-05-29T10:16:30Z","title":"GRKV: Global Regression for Training-Free KV Cache Compression in Long-Context LLMs","version":1},"reference_index":5,"source":"arxiv_source","source_observed_at":"2026-06-28T22:54:55.101568Z"},"links":{"cited_paper":"/paper/2407.21118","citing_paper":"/paper/2605.31105"},"observation_digest":"sha256:a0d74fb52b149930c9c5b476077507365019bd76ae76f223cdff1c2e7d13e28b","observation_id":"554a9a90-546c-4240-a260-d11ac706e328","resolution":{"observed_at":"2026-07-01T19:16:00.774202Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.21118","last_updated":"2024-11-04T02:08:55Z","snapshot_observed_at":"2026-08-12T23:11:46.275639Z","submitted_at":"2024-07-30T18:19:38Z","title":"Palu: Compressing KV-Cache with Low-Rank Projection","version":2},"cited_work":{"arxiv_id":"2407.21118","doi":null,"metadata_source":"pith","pith_arxiv_id":"2407.21118","snapshot_observed_at":"2026-07-10T01:36:44.121083Z","title":"Palu: Compressing kv-cache with low-rank projection.arXiv preprint arXiv:2407.21118","venue":"cs.AI","work_id":"dc6247ce-a2da-431d-a935-3fb13543cb13","year":2024},"citing_paper":{"arxiv_id":"2606.00535","last_updated":"2026-05-30T05:05:24Z","snapshot_observed_at":"2026-08-16T00:43:44.943332Z","submitted_at":"2026-05-30T05:05:24Z","title":"DREAM-S: Speculative Decoding with Searchable Drafting and Target-Aware Refinement for Multimodal Generation","version":1},"reference_index":56,"source":"arxiv_source","source_observed_at":"2026-06-28T18:55:51.474956Z"},"links":{"cited_paper":"/paper/2407.21118","citing_paper":"/paper/2606.00535"},"observation_digest":"sha256:3f2282702259194596d9acae937a29af9f8a24cfb0b93f9ca6934a06d0db1521","observation_id":"06d2376f-b0a2-4a6c-a291-d84b8421dc2c","resolution":{"observed_at":"2026-06-28T19:02:34.508413Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.21118","last_updated":"2024-11-04T02:08:55Z","snapshot_observed_at":"2026-08-12T23:11:46.275639Z","submitted_at":"2024-07-30T18:19:38Z","title":"Palu: Compressing KV-Cache with Low-Rank Projection","version":2},"cited_work":{"arxiv_id":"2407.21118","doi":null,"metadata_source":"pith","pith_arxiv_id":"2407.21118","snapshot_observed_at":"2026-07-10T01:36:44.121083Z","title":"Palu: Compressing kv-cache with low-rank projection.arXiv preprint arXiv:2407.21118","venue":"cs.AI","work_id":"dc6247ce-a2da-431d-a935-3fb13543cb13","year":2024},"citing_paper":{"arxiv_id":"2606.00573","last_updated":"2026-05-30T06:53:23Z","snapshot_observed_at":"2026-08-16T03:38:46.307266Z","submitted_at":"2026-05-30T06:53:23Z","title":"LASER: Loss-Aware Singular-value Decomposition and Rank Allocation for Efficient Low-Precision Vision-Language Models","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-06-28T19:09:29.347284Z"},"links":{"cited_paper":"/paper/2407.21118","citing_paper":"/paper/2606.00573"},"observation_digest":"sha256:c2b2755eaace6144c6e4beba7130921ae6f200204a71dc6958670ddb54f5a142","observation_id":"90c87189-d919-4472-a996-0f980e90c98a","resolution":{"observed_at":"2026-06-28T19:12:34.707197Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.21118","last_updated":"2024-11-04T02:08:55Z","snapshot_observed_at":"2026-08-12T23:11:46.275639Z","submitted_at":"2024-07-30T18:19:38Z","title":"Palu: Compressing KV-Cache with Low-Rank Projection","version":2},"cited_work":{"arxiv_id":"2407.21118","doi":null,"metadata_source":"pith","pith_arxiv_id":"2407.21118","snapshot_observed_at":"2026-07-10T01:36:44.121083Z","title":"Palu: Compressing kv-cache with low-rank projection.arXiv preprint arXiv:2407.21118","venue":"cs.AI","work_id":"dc6247ce-a2da-431d-a935-3fb13543cb13","year":2024},"citing_paper":{"arxiv_id":"2606.08565","last_updated":"2026-06-07T10:43:30Z","snapshot_observed_at":"2026-08-12T17:55:34.805021Z","submitted_at":"2026-06-07T10:43:30Z","title":"EinSort: Sorting is All We Need for Tensorizing LLM","version":1},"reference_index":14,"source":"arxiv_source","source_observed_at":"2026-06-27T18:31:01.804061Z"},"links":{"cited_paper":"/paper/2407.21118","citing_paper":"/paper/2606.08565"},"observation_digest":"sha256:f88e556db54e87c8bb7d780c3b9bdc4f0630b0ac7cb36669fa15e3db6d93f768","observation_id":"af4941be-b6f9-4ab6-80a6-1de7e319cc81","resolution":{"observed_at":"2026-07-02T22:57:26.699258Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.21118","last_updated":"2024-11-04T02:08:55Z","snapshot_observed_at":"2026-08-12T23:11:46.275639Z","submitted_at":"2024-07-30T18:19:38Z","title":"Palu: Compressing KV-Cache with Low-Rank Projection","version":2},"cited_work":{"arxiv_id":"2407.21118","doi":null,"metadata_source":"pith","pith_arxiv_id":"2407.21118","snapshot_observed_at":"2026-07-10T01:36:44.121083Z","title":"Palu: Compressing kv-cache with low-rank projection.arXiv preprint arXiv:2407.21118","venue":"cs.AI","work_id":"dc6247ce-a2da-431d-a935-3fb13543cb13","year":2024},"citing_paper":{"arxiv_id":"2606.11164","last_updated":"2026-06-09T17:44:23Z","snapshot_observed_at":"2026-08-14T09:16:22.038523Z","submitted_at":"2026-06-09T17:44:23Z","title":"ReasonAlloc: Hierarchical Decoding-Time KV Cache Budget Allocation for Reasoning Models","version":1},"reference_index":11,"source":"arxiv_source","source_observed_at":"2026-06-27T13:09:37.979051Z"},"links":{"cited_paper":"/paper/2407.21118","citing_paper":"/paper/2606.11164"},"observation_digest":"sha256:0b291a4f2f56ee4eef6c49fa6c9f37ef7d6376c54e7c74940e19fe77893b0522","observation_id":"69977081-c482-4a04-864e-82f935153758","resolution":{"observed_at":"2026-07-03T05:37:40.348156Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.21118","last_updated":"2024-11-04T02:08:55Z","snapshot_observed_at":"2026-08-12T23:11:46.275639Z","submitted_at":"2024-07-30T18:19:38Z","title":"Palu: Compressing KV-Cache with Low-Rank Projection","version":2},"cited_work":{"arxiv_id":"2407.21118","doi":null,"metadata_source":"pith","pith_arxiv_id":"2407.21118","snapshot_observed_at":"2026-07-10T01:36:44.121083Z","title":"Palu: Compressing kv-cache with low-rank projection.arXiv preprint arXiv:2407.21118","venue":"cs.AI","work_id":"dc6247ce-a2da-431d-a935-3fb13543cb13","year":2024},"citing_paper":{"arxiv_id":"2607.08032","last_updated":"2026-07-09T01:15:03Z","snapshot_observed_at":"2026-08-15T04:39:02.467469Z","submitted_at":"2026-07-09T01:15:03Z","title":"What to Keep, What to Forget: A Rate--Distortion View of Memory Compaction in LLMs and Agents","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-07-10T01:26:59.421158Z"},"links":{"cited_paper":"/paper/2407.21118","citing_paper":"/paper/2607.08032"},"observation_digest":"sha256:928083d639124eddcac359c6c06480ddeea878d619d4f3d8f59b69a34cac1567","observation_id":"ec9e31b7-358a-4438-9a9a-195ca299eee0","resolution":{"observed_at":"2026-07-10T01:36:44.122209Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-16T06:30:59.297886+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.21118","last_updated":"2024-11-04T02:08:55Z","snapshot_observed_at":"2026-08-12T23:11:46.275639Z","submitted_at":"2024-07-30T18:19:38Z","title":"Palu: Compressing KV-Cache with Low-Rank Projection","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.21118","snapshot_observed_at":"2026-08-02T06:31:50.083575Z","title":"Abdelfattah, and Kai-Chiang Wu","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.12550","last_updated":"2026-07-17T17:45:07Z","snapshot_observed_at":"2026-08-07T16:25:57.752265Z","submitted_at":"2026-07-14T09:23:20Z","title":"A JoLT for the KV Cache: Near-Lossless KV Cache Compression via Joint Tucker and JL-Residual Allocation for LLMs","version":2},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-02T06:31:50.083575Z"},"links":{"cited_paper":"/paper/2407.21118","citing_paper":"/paper/2607.12550"},"observation_digest":"sha256:617b2c6200fdee5f232dc7fdbf49ed9942d96c346df5768682d381f454b1073a","observation_id":"3e2573aa-faa8-4189-b696-6d4092c86a7b","resolution":{"observed_at":"2026-08-02T06:31:50.083575Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.21118","last_updated":"2024-11-04T02:08:55Z","snapshot_observed_at":"2026-08-12T23:11:46.275639Z","submitted_at":"2024-07-30T18:19:38Z","title":"Palu: Compressing KV-Cache with Low-Rank Projection","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.21118","snapshot_observed_at":"2026-08-02T06:50:14.941014Z","title":"Palu: Compressing kv-cache with low-rank projection.arXiv preprint arXiv:2407.21118, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.13088","last_updated":"2026-07-13T16:45:04Z","snapshot_observed_at":"2026-08-14T03:16:20.039338Z","submitted_at":"2026-07-13T16:45:04Z","title":"Securing LLMs in the Wild: Privacy and Security Challenges at the Edge","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-02T06:50:14.941014Z"},"links":{"cited_paper":"/paper/2407.21118","citing_paper":"/paper/2607.13088"},"observation_digest":"sha256:57f4d7e3c642f1000369ce82b43a17922e19795593bda16a4b1930f6de92a5b5","observation_id":"7c90bfd3-a5db-4c70-8f38-0575b9e87fb7","resolution":{"observed_at":"2026-08-02T06:50:14.941014Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.21118","last_updated":"2024-11-04T02:08:55Z","snapshot_observed_at":"2026-08-12T23:11:46.275639Z","submitted_at":"2024-07-30T18:19:38Z","title":"Palu: Compressing KV-Cache with Low-Rank Projection","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.21118","snapshot_observed_at":"2026-07-31T18:02:37.410810Z","title":"Palu: Compressing kv-cache with low-rank projection,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.24331","last_updated":"2026-07-27T12:08:56Z","snapshot_observed_at":"2026-08-13T09:00:54.655111Z","submitted_at":"2026-07-27T12:08:56Z","title":"DynaCalKV: Key-Value Cache Compression via Head Grouping and Adaptive Rank Allocation","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-07-31T18:02:37.410810Z"},"links":{"cited_paper":"/paper/2407.21118","citing_paper":"/paper/2607.24331"},"observation_digest":"sha256:5512c4ab03c1f705531ad25c946388761d4dd24fa7b974f9c7ae8062c7ddf300","observation_id":"d4431f5f-d32c-4e5f-9a5b-6ed319e2395e","resolution":{"observed_at":"2026-07-31T18:02:37.410810Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.21118","last_updated":"2024-11-04T02:08:55Z","snapshot_observed_at":"2026-08-12T23:11:46.275639Z","submitted_at":"2024-07-30T18:19:38Z","title":"Palu: Compressing KV-Cache with Low-Rank Projection","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.21118","snapshot_observed_at":"2026-07-31T11:49:11.614843Z","title":"Palu: Compressing KV-cache with low-rank projection","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.24555","last_updated":"2026-07-27T15:28:52Z","snapshot_observed_at":"2026-08-02T06:50:39.173138Z","submitted_at":"2026-07-27T15:28:52Z","title":"LOCKS: Page-Local Compact Key Summaries for Efficient Long-Context Decoding","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-07-31T11:49:11.614843Z"},"links":{"cited_paper":"/paper/2407.21118","citing_paper":"/paper/2607.24555"},"observation_digest":"sha256:697538c7f003b647df491dbbe6e7be0388319bb908a870c1ebe9adf37c28b7c5","observation_id":"9efb6cf7-0d03-46c7-bd1f-c19c47410427","resolution":{"observed_at":"2026-07-31T11:49:11.614843Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.21118","last_updated":"2024-11-04T02:08:55Z","snapshot_observed_at":"2026-08-12T23:11:46.275639Z","submitted_at":"2024-07-30T18:19:38Z","title":"Palu: Compressing KV-Cache with Low-Rank Projection","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.21118","snapshot_observed_at":"2026-08-03T00:43:48.033830Z","title":"Abdelfattah, and Kai-Chiang Wu","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.28699","last_updated":"2026-08-06T10:44:15Z","snapshot_observed_at":"2026-08-14T06:51:28.858675Z","submitted_at":"2026-07-30T11:04:45Z","title":"WitCert: Sound Runtime Risk Observability and Gating for KV-Cache Quantization","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-03T00:43:48.033830Z"},"links":{"cited_paper":"/paper/2407.21118","citing_paper":"/paper/2607.28699"},"observation_digest":"sha256:eab106cb1c738ff2607cfbab24f782a40d66ee065a8b0324bfa37adc6d502fda","observation_id":"0cefdd22-aad1-4e4c-80c1-cb14334d7f0d","resolution":{"observed_at":"2026-08-03T00:43:48.033830Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.21118","last_updated":"2024-11-04T02:08:55Z","snapshot_observed_at":"2026-08-12T23:11:46.275639Z","submitted_at":"2024-07-30T18:19:38Z","title":"Palu: Compressing KV-Cache with Low-Rank Projection","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.21118","snapshot_observed_at":"2026-08-04T07:49:39.405147Z","title":"arXiv preprint arXiv:2407.21118 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.02415","last_updated":"2026-08-03T15:53:49Z","snapshot_observed_at":"2026-08-11T21:43:49.975246Z","submitted_at":"2026-08-03T15:53:49Z","title":"Training-Free versus Training-Based Intent Classification in LLMs: Accuracy, Robustness, and Failure Modes","version":1},"reference_index":50,"source":"arxiv_source","source_observed_at":"2026-08-04T07:49:39.405147Z"},"links":{"cited_paper":"/paper/2407.21118","citing_paper":"/paper/2608.02415"},"observation_digest":"sha256:9ef173af6e2e6150d01fb3b981b8aab706d0cdd7b7986cf0a8742c9ac3780dd6","observation_id":"cfef5d0d-dce5-4308-b32a-06239cc816f5","resolution":{"observed_at":"2026-08-04T07:49:39.405147Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.21118","last_updated":"2024-11-04T02:08:55Z","snapshot_observed_at":"2026-08-12T23:11:46.275639Z","submitted_at":"2024-07-30T18:19:38Z","title":"Palu: Compressing KV-Cache with Low-Rank Projection","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.21118","snapshot_observed_at":"2026-08-15T15:02:52.922607Z","title":"Palu: Compressing KV-Cache with Low-Rank Projection , journal =","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.02901","last_updated":"2026-08-03T21:38:30Z","snapshot_observed_at":"2026-08-15T17:14:49.606481Z","submitted_at":"2026-08-03T21:38:30Z","title":"AnchorKV: Anchor-Residual KV Cache Compression","version":1},"reference_index":30,"source":"arxiv_source","source_observed_at":"2026-08-15T15:02:52.922607Z"},"links":{"cited_paper":"/paper/2407.21118","citing_paper":"/paper/2608.02901"},"observation_digest":"sha256:1ddb41684561d235e898cd52f8f9b026f41ce7bf0d3f85122b6ed206532ac486","observation_id":"064a8211-11ca-45ae-b1ce-7bd7c3b7bc12","resolution":{"observed_at":"2026-08-15T15:02:52.922607Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.21118","last_updated":"2024-11-04T02:08:55Z","snapshot_observed_at":"2026-08-12T23:11:46.275639Z","submitted_at":"2024-07-30T18:19:38Z","title":"Palu: Compressing KV-Cache with Low-Rank Projection","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.21118","snapshot_observed_at":"2026-08-05T23:10:35.285100Z","title":"Palu: Compressing KV-Cache with Low-Rank Projection","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.03228","last_updated":"2026-08-05T20:50:42Z","snapshot_observed_at":"2026-08-14T09:50:26.014755Z","submitted_at":"2026-08-04T06:59:29Z","title":"SAKI: Score-Aware Low-Rank Key Indexing with Random-Matrix Noise Correction for KV Retrieval","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-05T23:10:35.285100Z"},"links":{"cited_paper":"/paper/2407.21118","citing_paper":"/paper/2608.03228"},"observation_digest":"sha256:7281cd0265fa287b315b12d89d7e56f58303d41de9da3269647d004b21654c07","observation_id":"d3be1814-e7c1-4d18-aa49-b2d1acf584b7","resolution":{"observed_at":"2026-08-05T23:10:35.285100Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.21118","last_updated":"2024-11-04T02:08:55Z","snapshot_observed_at":"2026-08-12T23:11:46.275639Z","submitted_at":"2024-07-30T18:19:38Z","title":"Palu: Compressing KV-Cache with Low-Rank Projection","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.21118","snapshot_observed_at":"2026-08-08T00:54:42.694725Z","title":"Palu: Compressing KV-Cache with Low-Rank Projection","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.03228","last_updated":"2026-08-05T20:50:42Z","snapshot_observed_at":"2026-08-14T09:50:26.014755Z","submitted_at":"2026-08-04T06:59:29Z","title":"SAKI: Score-Aware Low-Rank Key Indexing with Random-Matrix Noise Correction for KV Retrieval","version":2},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-08T00:54:42.694725Z"},"links":{"cited_paper":"/paper/2407.21118","citing_paper":"/paper/2608.03228"},"observation_digest":"sha256:b1da49f83691a1d8f2a6947e7d1e39aa965813437040c12f841526791fd81691","observation_id":"b4baf7a2-63fd-442c-ba6c-a225c61f16dc","resolution":{"observed_at":"2026-08-08T00:54:42.694725Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.21118","last_updated":"2024-11-04T02:08:55Z","snapshot_observed_at":"2026-08-12T23:11:46.275639Z","submitted_at":"2024-07-30T18:19:38Z","title":"Palu: Compressing KV-Cache with Low-Rank Projection","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.21118","snapshot_observed_at":"2026-08-15T14:46:03.872365Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.04428","last_updated":"2026-08-05T04:17:03Z","snapshot_observed_at":"2026-08-16T01:42:08.131327Z","submitted_at":"2026-08-05T04:17:03Z","title":"Deltoris: Enabling Real-time VLA Inference in Embodied AI via Bit-level Sparsity and Speculative Inference","version":1},"reference_index":2024,"source":"pdf_text","source_observed_at":"2026-08-15T14:46:03.872365Z"},"links":{"cited_paper":"/paper/2407.21118","citing_paper":"/paper/2608.04428"},"observation_digest":"sha256:218fcae916c32e0f6ece3ba6f71c7068309ac6e34f853d51bd3e708caedd0d0f","observation_id":"ebec126e-da94-419a-a482-322eae95ddbb","resolution":{"observed_at":"2026-08-15T14:46:03.872365Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2407.21118/citation-record","integrity":"/paper/2407.21118/integrity","json":"/paper/2407.21118/citation-record.json","paper":"/paper/2407.21118"},"outbound":[],"paper":{"arxiv_id":"2407.21118","last_updated":"2024-11-04T02:08:55Z","latest_version":2,"primary_category":"cs.AI","snapshot_observed_at":"2026-08-12T23:11:46.275639Z","submitted_at":"2024-07-30T18:19:38Z","title":"Palu: Compressing KV-Cache with Low-Rank Projection"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-16T06:30:59.297886+00:00","source":"crossref"},{"observed_at":"2026-08-16T06:30:54.164669+00:00","source":"retraction_watch"}],"thesis":"As of 16 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 37 inbound Pith citation observations for arXiv:2407.21118."}