{"as_of":"2026-08-21T15:39:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:6aae61db2c56b0d9617fa3d35d75483f4010b61422708a150354dced9a2d44b9","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":30,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":30,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-21T06:32:19.484+00:00","state":"measured"},{"denominator":30,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":30,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-12T17:22:46.999943Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":1,"source":"pith","source_observed_at":"2026-08-05T02:28:24.338817Z","state":"measured"}],"external_citation_measurements":[{"count":2,"observed_at":"2026-08-05T02:28:24.338817Z","source":"pith"}],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2405.12981","last_updated":"2024-05-21T17:59:29Z","snapshot_observed_at":"2026-08-20T10:03:23.466960Z","submitted_at":"2024-05-21T17:59:29Z","title":"Reducing Transformer Key-Value Cache Size with Cross-Layer Attention","version":1},"cited_work":{"arxiv_id":"2405.12981","doi":"10.48550/arxiv.2405.12981","metadata_source":"pith","pith_arxiv_id":"2405.12981","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Reducing transformer key-value cache size with cross-layer attention","venue":"cs.LG","work_id":"e92a5a41-6fe4-4a0f-84a0-b866c6b353f3","year":2024},"citing_paper":{"arxiv_id":"2410.13846","last_updated":"2026-05-18T05:12:14Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-10-17T17:58:14Z","title":"LightTransfer: Your Long-Context LLM is Secretly a Hybrid Model with Effortless Adaptation","version":3},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-05-23T18:31:35.391674Z"},"links":{"cited_paper":"/paper/2405.12981","citing_paper":"/paper/2410.13846"},"observation_digest":"sha256:829c4eb06370b44328d828a3c3036b1d6ab884206ac3a42426fa37891f34bfe6","observation_id":"c00b63c8-c9a6-4544-978d-b68ed14a0472","resolution":{"observed_at":"2026-05-23T18:33:19.446287Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.12981","last_updated":"2024-05-21T17:59:29Z","snapshot_observed_at":"2026-08-20T10:03:23.466960Z","submitted_at":"2024-05-21T17:59:29Z","title":"Reducing Transformer Key-Value Cache Size with Cross-Layer Attention","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.12981","snapshot_observed_at":"2026-08-12T17:22:46.999943Z","title":"arXiv preprint arXiv:2405.12981 (2024) 2","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2411.12663","last_updated":"2024-11-19T17:16:31Z","snapshot_observed_at":"2026-08-19T03:20:46.539842Z","submitted_at":"2024-11-19T17:16:31Z","title":"PoM: Efficient Image and Video Generation with the Polynomial Mixer","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-12T17:22:46.999943Z"},"links":{"cited_paper":"/paper/2405.12981","citing_paper":"/paper/2411.12663"},"observation_digest":"sha256:9d5be972f960b3d80fad2f6d00c6c5286a2fd009f2df183fecc6e60d573f2f1f","observation_id":"ec66c560-a6b2-4182-a552-453dd9e2a2e0","resolution":{"observed_at":"2026-08-12T17:22:46.999943Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.12981","last_updated":"2024-05-21T17:59:29Z","snapshot_observed_at":"2026-08-20T10:03:23.466960Z","submitted_at":"2024-05-21T17:59:29Z","title":"Reducing Transformer Key-Value Cache Size with Cross-Layer Attention","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.12981","snapshot_observed_at":"2026-08-12T16:20:32.344744Z","title":"Reducing transformer key-value cache size with cross-layer attention","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2411.13676","last_updated":"2024-11-20T19:51:25Z","snapshot_observed_at":"2026-08-14T20:26:44.533985Z","submitted_at":"2024-11-20T19:51:25Z","title":"Hymba: A Hybrid-head Architecture for Small Language Models","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-12T16:20:32.344744Z"},"links":{"cited_paper":"/paper/2405.12981","citing_paper":"/paper/2411.13676"},"observation_digest":"sha256:bed026e3bef6e98a426c137e418949e6d84d60be668c2fa3d1d51ea67c7e5139","observation_id":"baa5f6ec-1d50-4f0b-9471-2fe7120524d9","resolution":{"observed_at":"2026-08-12T16:20:32.344744Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.12981","last_updated":"2024-05-21T17:59:29Z","snapshot_observed_at":"2026-08-20T10:03:23.466960Z","submitted_at":"2024-05-21T17:59:29Z","title":"Reducing Transformer Key-Value Cache Size with Cross-Layer Attention","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.12981","snapshot_observed_at":"2026-08-12T12:13:55.149273Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2411.17426","last_updated":"2025-01-31T14:13:49Z","snapshot_observed_at":"2026-08-19T07:19:32.477239Z","submitted_at":"2024-11-26T13:34:02Z","title":"CLOVER: Cross-Layer Orthogonal Vectors Pruning and Fine-Tuning","version":3},"reference_index":2020,"source":"pdf_text","source_observed_at":"2026-08-12T12:13:55.149273Z"},"links":{"cited_paper":"/paper/2405.12981","citing_paper":"/paper/2411.17426"},"observation_digest":"sha256:b933cd1797f61ea536cc203534e8e58cd9ade71eb2a1ef981e81986ffdbef244","observation_id":"24e3e211-2a95-4ad7-ae8a-add9f701ff2f","resolution":{"observed_at":"2026-08-12T12:13:55.149273Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.12981","last_updated":"2024-05-21T17:59:29Z","snapshot_observed_at":"2026-08-20T10:03:23.466960Z","submitted_at":"2024-05-21T17:59:29Z","title":"Reducing Transformer Key-Value Cache Size with Cross-Layer Attention","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.12981","snapshot_observed_at":"2026-08-12T11:37:07.553067Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2411.18077","last_updated":"2025-06-08T21:23:22Z","snapshot_observed_at":"2026-08-15T19:30:15.609643Z","submitted_at":"2024-11-27T06:10:49Z","title":"MiniKV: Pushing the Limits of LLM Inference via 2-Bit Layer-Discriminative KV Cache","version":3},"reference_index":3,"source":"arxiv_source","source_observed_at":"2026-08-12T11:37:07.553067Z"},"links":{"cited_paper":"/paper/2405.12981","citing_paper":"/paper/2411.18077"},"observation_digest":"sha256:fd880c5f10a24dd208dbaae0f0ba3b81794dc0597e8aec619dcc5c85198479db","observation_id":"592c3a29-2acc-4656-848d-50800257f37c","resolution":{"observed_at":"2026-08-12T11:37:07.553067Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.12981","last_updated":"2024-05-21T17:59:29Z","snapshot_observed_at":"2026-08-20T10:03:23.466960Z","submitted_at":"2024-05-21T17:59:29Z","title":"Reducing Transformer Key-Value Cache Size with Cross-Layer Attention","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.12981","snapshot_observed_at":"2026-08-11T23:46:01.984234Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.02252","last_updated":"2025-08-04T02:17:56Z","snapshot_observed_at":"2026-08-16T15:44:22.755394Z","submitted_at":"2024-12-03T08:29:27Z","title":"Compressing KV Cache for Long-Context LLM Inference with Inter-Layer Attention Similarity","version":2},"reference_index":8,"source":"arxiv_source","source_observed_at":"2026-08-11T23:46:01.984234Z"},"links":{"cited_paper":"/paper/2405.12981","citing_paper":"/paper/2412.02252"},"observation_digest":"sha256:5e0561a23ace693dc751f7baa85ef7006802232515268142f692e30a3bf536b2","observation_id":"87ff2fb3-099b-47c4-8c1a-56e45b31dc06","resolution":{"observed_at":"2026-08-11T23:46:01.984234Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.12981","last_updated":"2024-05-21T17:59:29Z","snapshot_observed_at":"2026-08-20T10:03:23.466960Z","submitted_at":"2024-05-21T17:59:29Z","title":"Reducing Transformer Key-Value Cache Size with Cross-Layer Attention","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.12981","snapshot_observed_at":"2026-08-11T23:37:44.871522Z","title":"Reducing transformer key-value cache size with cross-layer attention","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.02344","last_updated":"2024-12-03T10:04:15Z","snapshot_observed_at":"2026-08-20T22:54:05.239556Z","submitted_at":"2024-12-03T10:04:15Z","title":"UniForm: A Reuse Attention Mechanism Optimized for Efficient Vision Transformers on Edge Devices","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-11T23:37:44.871522Z"},"links":{"cited_paper":"/paper/2405.12981","citing_paper":"/paper/2412.02344"},"observation_digest":"sha256:ab47d6e1f4fd49ab94fd331315d4d817a58b8e19a7ab6a3fa38e2c5b9e8ba95f","observation_id":"a0785f33-eacf-4f05-ab0a-3c3e6e476661","resolution":{"observed_at":"2026-08-11T23:37:44.871522Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.12981","last_updated":"2024-05-21T17:59:29Z","snapshot_observed_at":"2026-08-20T10:03:23.466960Z","submitted_at":"2024-05-21T17:59:29Z","title":"Reducing Transformer Key-Value Cache Size with Cross-Layer Attention","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.12981","snapshot_observed_at":"2026-08-11T13:52:48.622944Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.12706","last_updated":"2025-02-20T12:14:49Z","snapshot_observed_at":"2026-08-13T11:09:26.001382Z","submitted_at":"2024-12-17T09:20:31Z","title":"More Tokens, Lower Precision: Towards the Optimal Token-Precision Trade-off in KV Cache Compression","version":2},"reference_index":5,"source":"arxiv_source","source_observed_at":"2026-08-11T13:52:48.622944Z"},"links":{"cited_paper":"/paper/2405.12981","citing_paper":"/paper/2412.12706"},"observation_digest":"sha256:460c29925a64e781485d5ced92cdd1ec7c0fd947c3c89e9770acd3d46a286e44","observation_id":"3572522b-9059-4164-93e1-a936006ebe83","resolution":{"observed_at":"2026-08-11T13:52:48.622944Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.12981","last_updated":"2024-05-21T17:59:29Z","snapshot_observed_at":"2026-08-20T10:03:23.466960Z","submitted_at":"2024-05-21T17:59:29Z","title":"Reducing Transformer Key-Value Cache Size with Cross-Layer Attention","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.12981","snapshot_observed_at":"2026-08-11T11:57:17.878098Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.14838","last_updated":"2025-05-27T03:08:57Z","snapshot_observed_at":"2026-08-21T02:11:05.377768Z","submitted_at":"2024-12-19T13:28:42Z","title":"DynamicKV: Task-Aware Adaptive KV Cache Compression for Long Context LLMs","version":4},"reference_index":3,"source":"arxiv_source","source_observed_at":"2026-08-11T11:57:17.878098Z"},"links":{"cited_paper":"/paper/2405.12981","citing_paper":"/paper/2412.14838"},"observation_digest":"sha256:8f5b4639933e7c051fec8971d75cbdd9a77d2f7f62cc58a11f3139d6a480d556","observation_id":"a5c07440-d1ed-432c-a602-7b2ea45b4c59","resolution":{"observed_at":"2026-08-11T11:57:17.878098Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.12981","last_updated":"2024-05-21T17:59:29Z","snapshot_observed_at":"2026-08-20T10:03:23.466960Z","submitted_at":"2024-05-21T17:59:29Z","title":"Reducing Transformer Key-Value Cache Size with Cross-Layer Attention","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.12981","snapshot_observed_at":"2026-08-11T05:32:29.385894Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.17483","last_updated":"2024-12-23T11:24:04Z","snapshot_observed_at":"2026-08-17T18:48:05.903830Z","submitted_at":"2024-12-23T11:24:04Z","title":"A Silver Bullet or a Compromise for Full Attention? A Comprehensive Study of Gist Token-based Context Compression","version":1},"reference_index":3,"source":"arxiv_source","source_observed_at":"2026-08-11T05:32:29.385894Z"},"links":{"cited_paper":"/paper/2405.12981","citing_paper":"/paper/2412.17483"},"observation_digest":"sha256:256aa075f03459a336c74791e3acba34a68d8f4143d054a227c924a03a1ceff4","observation_id":"357356ce-20ee-44fa-b58b-cab024cf6828","resolution":{"observed_at":"2026-08-11T05:32:29.385894Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.12981","last_updated":"2024-05-21T17:59:29Z","snapshot_observed_at":"2026-08-20T10:03:23.466960Z","submitted_at":"2024-05-21T17:59:29Z","title":"Reducing Transformer Key-Value Cache Size with Cross-Layer Attention","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.12981","snapshot_observed_at":"2026-08-11T00:55:03.777756Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.19255","last_updated":"2025-01-14T05:48:07Z","snapshot_observed_at":"2026-08-14T11:12:42.419587Z","submitted_at":"2024-12-26T15:45:45Z","title":"Multi-matrix Factorization Attention","version":2},"reference_index":6,"source":"arxiv_source","source_observed_at":"2026-08-11T00:55:03.777756Z"},"links":{"cited_paper":"/paper/2405.12981","citing_paper":"/paper/2412.19255"},"observation_digest":"sha256:a5d6e38b6a090c8d8080761e687b5d69d420ab899f40ad54e3a1f25aeb987cbf","observation_id":"e6d12d0e-1a79-4b0e-8f87-f82d9fe68267","resolution":{"observed_at":"2026-08-11T00:55:03.777756Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.12981","last_updated":"2024-05-21T17:59:29Z","snapshot_observed_at":"2026-08-20T10:03:23.466960Z","submitted_at":"2024-05-21T17:59:29Z","title":"Reducing Transformer Key-Value Cache Size with Cross-Layer Attention","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.12981","snapshot_observed_at":"2026-08-11T00:38:48.814000Z","title":"Reducing Transformer Key-Value Cache Size with Cross-Layer Attention,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.19442","last_updated":"2025-07-30T05:24:46Z","snapshot_observed_at":"2026-08-20T14:01:46.841630Z","submitted_at":"2024-12-27T04:17:57Z","title":"A Survey on Large Language Model Acceleration based on KV Cache Management","version":3},"reference_index":205,"source":"pdf_text","source_observed_at":"2026-08-11T00:38:48.814000Z"},"links":{"cited_paper":"/paper/2405.12981","citing_paper":"/paper/2412.19442"},"observation_digest":"sha256:46e9821148841dbd055950ced844fdcc496e813aecc4080b7f4d1e3e124a07f1","observation_id":"2cc63f34-a742-404f-b05d-86d8dcaa63a1","resolution":{"observed_at":"2026-08-11T00:38:48.814000Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.12981","last_updated":"2024-05-21T17:59:29Z","snapshot_observed_at":"2026-08-20T10:03:23.466960Z","submitted_at":"2024-05-21T17:59:29Z","title":"Reducing Transformer Key-Value Cache Size with Cross-Layer Attention","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.12981","snapshot_observed_at":"2026-08-10T20:14:58.581590Z","title":"Reducing transformer key-value cache size with cross-layer attention","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2501.09223","last_updated":"2025-06-15T13:24:11Z","snapshot_observed_at":"2026-08-15T10:26:22.406164Z","submitted_at":"2025-01-16T01:03:56Z","title":"Foundations of Large Language Models","version":2},"reference_index":22,"source":"arxiv_source","source_observed_at":"2026-08-10T20:14:58.581590Z"},"links":{"cited_paper":"/paper/2405.12981","citing_paper":"/paper/2501.09223"},"observation_digest":"sha256:ae698168d755f76f96b14c12c590a2e4a502a05299092f8ded2563c03d95dad9","observation_id":"c0982b95-36e1-4cb1-8f07-7655903b600f","resolution":{"observed_at":"2026-08-10T20:14:58.581590Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.12981","last_updated":"2024-05-21T17:59:29Z","snapshot_observed_at":"2026-08-20T10:03:23.466960Z","submitted_at":"2024-05-21T17:59:29Z","title":"Reducing Transformer Key-Value Cache Size with Cross-Layer Attention","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.12981","snapshot_observed_at":"2026-08-10T15:50:21.846487Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2501.13629","last_updated":"2025-02-10T17:19:21Z","snapshot_observed_at":"2026-08-16T15:30:42.801825Z","submitted_at":"2025-01-23T12:58:14Z","title":"Sigma: Differential Rescaling of Query, Key and Value for Efficient Language Models","version":2},"reference_index":2020,"source":"pdf_text","source_observed_at":"2026-08-10T15:50:21.846487Z"},"links":{"cited_paper":"/paper/2405.12981","citing_paper":"/paper/2501.13629"},"observation_digest":"sha256:581a8eefc2ae05acd7a5560d5a537bef222e1474d58e29dcc215da9a0b33ce59","observation_id":"2a8512ca-d088-4f2a-b192-7d6577b5871b","resolution":{"observed_at":"2026-08-10T15:50:21.846487Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.12981","last_updated":"2024-05-21T17:59:29Z","snapshot_observed_at":"2026-08-20T10:03:23.466960Z","submitted_at":"2024-05-21T17:59:29Z","title":"Reducing Transformer Key-Value Cache Size with Cross-Layer Attention","version":1},"cited_work":{"arxiv_id":"2405.12981","doi":"10.48550/arxiv.2405.12981","metadata_source":"pith","pith_arxiv_id":"2405.12981","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Reducing transformer key-value cache size with cross-layer attention","venue":"cs.LG","work_id":"e92a5a41-6fe4-4a0f-84a0-b866c6b353f3","year":2024},"citing_paper":{"arxiv_id":"2502.01941","last_updated":"2026-05-12T08:04:27Z","snapshot_observed_at":"2026-08-14T08:23:35.104382Z","submitted_at":"2025-02-04T02:23:06Z","title":"Semantic Integrity Matters: Benchmarking and Preserving High-Density Reasoning in KV Cache Compression","version":4},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-05-23T04:15:36.906263Z"},"links":{"cited_paper":"/paper/2405.12981","citing_paper":"/paper/2502.01941"},"observation_digest":"sha256:4dc73046f319720786766bde0a94fcf104e02ae5141edf132980b52ae6671721","observation_id":"60eeb619-40af-4ecb-9862-e62d815b8235","resolution":{"observed_at":"2026-05-23T04:17:31.245179Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.12981","last_updated":"2024-05-21T17:59:29Z","snapshot_observed_at":"2026-08-20T10:03:23.466960Z","submitted_at":"2024-05-21T17:59:29Z","title":"Reducing Transformer Key-Value Cache Size with Cross-Layer Attention","version":1},"cited_work":{"arxiv_id":"2405.12981","doi":"10.48550/arxiv.2405.12981","metadata_source":"pith","pith_arxiv_id":"2405.12981","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Reducing transformer key-value cache size with cross-layer attention","venue":"cs.LG","work_id":"e92a5a41-6fe4-4a0f-84a0-b866c6b353f3","year":2024},"citing_paper":{"arxiv_id":"2502.05171","last_updated":"2025-02-17T17:14:04Z","snapshot_observed_at":"2026-08-15T00:53:48.099058Z","submitted_at":"2025-02-07T18:55:02Z","title":"Scaling up Test-Time Compute with Latent Reasoning: A Recurrent Depth Approach","version":2},"reference_index":26,"source":"arxiv_source","source_observed_at":"2026-05-12T15:39:40.845703Z"},"links":{"cited_paper":"/paper/2405.12981","citing_paper":"/paper/2502.05171"},"observation_digest":"sha256:9409fefa01325bbd0418abc1e814e3e011d87ae18a7df76deff7b21b2fb13245","observation_id":"1db8d200-bea1-4583-a4f8-07dca902121d","resolution":{"observed_at":"2026-05-12T15:39:41.169681Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.12981","last_updated":"2024-05-21T17:59:29Z","snapshot_observed_at":"2026-08-20T10:03:23.466960Z","submitted_at":"2024-05-21T17:59:29Z","title":"Reducing Transformer Key-Value Cache Size with Cross-Layer Attention","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.12981","snapshot_observed_at":"2026-08-08T14:13:37.965936Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.06975","last_updated":"2025-02-10T19:14:51Z","snapshot_observed_at":"2026-08-14T08:32:12.525435Z","submitted_at":"2025-02-10T19:14:51Z","title":"Position: Episodic Memory is the Missing Piece for Long-Term LLM Agents","version":1},"reference_index":10,"source":"arxiv_source","source_observed_at":"2026-08-08T14:13:37.965936Z"},"links":{"cited_paper":"/paper/2405.12981","citing_paper":"/paper/2502.06975"},"observation_digest":"sha256:c64e713d2156a95e92aebbb40057f2dbbfb54111d594cfdbe263a53bcd589102","observation_id":"d4937893-d0a2-45b9-8c82-11ad5be1b371","resolution":{"observed_at":"2026-08-08T14:13:37.965936Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.12981","last_updated":"2024-05-21T17:59:29Z","snapshot_observed_at":"2026-08-20T10:03:23.466960Z","submitted_at":"2024-05-21T17:59:29Z","title":"Reducing Transformer Key-Value Cache Size with Cross-Layer Attention","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.12981","snapshot_observed_at":"2026-08-09T04:28:03.903083Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2502.10424","last_updated":"2025-02-05T20:43:48Z","snapshot_observed_at":"2026-08-20T23:39:16.252274Z","submitted_at":"2025-02-05T20:43:48Z","title":"QuantSpec: Self-Speculative Decoding with Hierarchical Quantized KV Cache","version":1},"reference_index":2024,"source":"pdf_text","source_observed_at":"2026-08-09T04:28:03.903083Z"},"links":{"cited_paper":"/paper/2405.12981","citing_paper":"/paper/2502.10424"},"observation_digest":"sha256:0b4309de63db900d4a389695744618cd6644e2ab0b2a5ad31f937351f4c4b045","observation_id":"30d4b816-ebab-44fb-b757-ee0633875c68","resolution":{"observed_at":"2026-08-09T04:28:03.903083Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.12981","last_updated":"2024-05-21T17:59:29Z","snapshot_observed_at":"2026-08-20T10:03:23.466960Z","submitted_at":"2024-05-21T17:59:29Z","title":"Reducing Transformer Key-Value Cache Size with Cross-Layer Attention","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.12981","snapshot_observed_at":"2026-08-07T22:34:07.961504Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.12170","last_updated":"2025-05-28T12:57:47Z","snapshot_observed_at":"2026-08-17T11:40:15.102906Z","submitted_at":"2025-02-13T10:26:27Z","title":"MUDDFormer: Breaking Residual Bottlenecks in Transformers via Multiway Dynamic Dense Connections","version":2},"reference_index":4,"source":"arxiv_source","source_observed_at":"2026-08-07T22:34:07.961504Z"},"links":{"cited_paper":"/paper/2405.12981","citing_paper":"/paper/2502.12170"},"observation_digest":"sha256:931054d8d5b16e345f7fbfbcb241a99963bdde8316c6eb5ea19f0d0990f982dc","observation_id":"9bbe0a22-fc85-4ac7-81a7-82c15fcbdcf7","resolution":{"observed_at":"2026-08-07T22:34:07.961504Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.12981","last_updated":"2024-05-21T17:59:29Z","snapshot_observed_at":"2026-08-20T10:03:23.466960Z","submitted_at":"2024-05-21T17:59:29Z","title":"Reducing Transformer Key-Value Cache Size with Cross-Layer Attention","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.12981","snapshot_observed_at":"2026-08-07T13:32:29.855424Z","title":"Reducing transformer key-value cache size with cross-layer attention, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.21487","last_updated":"2025-05-27T17:54:07Z","snapshot_observed_at":"2026-08-07T13:24:37.243429Z","submitted_at":"2025-05-27T17:54:07Z","title":"Hardware-Efficient Attention for Fast Decoding","version":1},"reference_index":10,"source":"arxiv_source","source_observed_at":"2026-08-07T13:32:29.855424Z"},"links":{"cited_paper":"/paper/2405.12981","citing_paper":"/paper/2505.21487"},"observation_digest":"sha256:2b36d024a39ed9a01d519465a907da89ec69c5cb7f285b4dfcc59ec90e321095","observation_id":"cd97bfa7-81dd-4579-b112-b259b2a85479","resolution":{"observed_at":"2026-08-07T13:32:29.855424Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.12981","last_updated":"2024-05-21T17:59:29Z","snapshot_observed_at":"2026-08-20T10:03:23.466960Z","submitted_at":"2024-05-21T17:59:29Z","title":"Reducing Transformer Key-Value Cache Size with Cross-Layer Attention","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.12981","snapshot_observed_at":"2026-08-06T17:20:25.790122Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.11273","last_updated":"2025-07-15T12:52:12Z","snapshot_observed_at":"2026-08-13T18:35:52.333072Z","submitted_at":"2025-07-15T12:52:12Z","title":"KV-Latent: Dimensional-level KV Cache Reduction with Frequency-aware Rotary Positional Embedding","version":1},"reference_index":7,"source":"arxiv_source","source_observed_at":"2026-08-06T17:20:25.790122Z"},"links":{"cited_paper":"/paper/2405.12981","citing_paper":"/paper/2507.11273"},"observation_digest":"sha256:24106a65cce89dac61c888e3c2e38d4035daffb1bc4bed284968d1769da668d5","observation_id":"b6a8c350-03bf-426b-891a-a5d23d81a1fd","resolution":{"observed_at":"2026-08-06T17:20:25.790122Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.12981","last_updated":"2024-05-21T17:59:29Z","snapshot_observed_at":"2026-08-20T10:03:23.466960Z","submitted_at":"2024-05-21T17:59:29Z","title":"Reducing Transformer Key-Value Cache Size with Cross-Layer Attention","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.12981","snapshot_observed_at":"2026-08-06T13:57:15.661526Z","title":"Reducing transformer key-value cache size with cross-layer attention","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.19906","last_updated":"2025-08-04T08:19:26Z","snapshot_observed_at":"2026-08-19T22:54:51.852475Z","submitted_at":"2025-07-26T10:34:53Z","title":"CaliDrop: KV Cache Compression with Calibration","version":2},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-06T13:57:15.661526Z"},"links":{"cited_paper":"/paper/2405.12981","citing_paper":"/paper/2507.19906"},"observation_digest":"sha256:d4c7781d923cdf0c57636fcc86830bed6be080bd3a1ee8505f2a43d619a78fe2","observation_id":"345bb7a4-d061-4786-9478-af7175703411","resolution":{"observed_at":"2026-08-06T13:57:15.661526Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.12981","last_updated":"2024-05-21T17:59:29Z","snapshot_observed_at":"2026-08-20T10:03:23.466960Z","submitted_at":"2024-05-21T17:59:29Z","title":"Reducing Transformer Key-Value Cache Size with Cross-Layer Attention","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.12981","snapshot_observed_at":"2026-08-05T17:55:09.661574Z","title":"Reducing transformer key-value cache size with cross-layer attention","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.15881","last_updated":"2025-08-25T02:24:20Z","snapshot_observed_at":"2026-08-19T17:54:54.763717Z","submitted_at":"2025-08-21T15:25:40Z","title":"TPLA: Tensor Parallel Latent Attention for Efficient Disaggregated Prefill and Decode Inference","version":2},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-05T17:55:09.661574Z"},"links":{"cited_paper":"/paper/2405.12981","citing_paper":"/paper/2508.15881"},"observation_digest":"sha256:039c4ef224296f1fd207d669df61ad91e96a86f2424f790639d809bda036aaec","observation_id":"6d5022ab-4df1-4ddb-b02e-098ef4e89e67","resolution":{"observed_at":"2026-08-05T17:55:09.661574Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.12981","last_updated":"2024-05-21T17:59:29Z","snapshot_observed_at":"2026-08-20T10:03:23.466960Z","submitted_at":"2024-05-21T17:59:29Z","title":"Reducing Transformer Key-Value Cache Size with Cross-Layer Attention","version":1},"cited_work":{"arxiv_id":"2405.12981","doi":"10.48550/arxiv.2405.12981","metadata_source":"pith","pith_arxiv_id":"2405.12981","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Reducing transformer key-value cache size with cross-layer attention","venue":"cs.LG","work_id":"e92a5a41-6fe4-4a0f-84a0-b866c6b353f3","year":2024},"citing_paper":{"arxiv_id":"2604.22782","last_updated":"2026-04-03T14:56:17Z","snapshot_observed_at":"2026-08-14T05:37:02.618322Z","submitted_at":"2026-04-03T14:56:17Z","title":"Stochastic KV Routing: Enabling Adaptive Depth-Wise Cache Sharing","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-05-13T19:56:48.015363Z"},"links":{"cited_paper":"/paper/2405.12981","citing_paper":"/paper/2604.22782"},"observation_digest":"sha256:288edabe3a38c7840f4ea5d6ed05f6c5523ad4c693ea39c7673c623e34327a91","observation_id":"0afe267d-674f-4f95-8d75-5f91593e0c76","resolution":{"observed_at":"2026-05-13T19:58:12.051857Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.12981","last_updated":"2024-05-21T17:59:29Z","snapshot_observed_at":"2026-08-20T10:03:23.466960Z","submitted_at":"2024-05-21T17:59:29Z","title":"Reducing Transformer Key-Value Cache Size with Cross-Layer Attention","version":1},"cited_work":{"arxiv_id":"2405.12981","doi":"10.48550/arxiv.2405.12981","metadata_source":"pith","pith_arxiv_id":"2405.12981","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Reducing transformer key-value cache size with cross-layer attention","venue":"cs.LG","work_id":"e92a5a41-6fe4-4a0f-84a0-b866c6b353f3","year":2024},"citing_paper":{"arxiv_id":"2605.07721","last_updated":"2026-05-19T08:08:32Z","snapshot_observed_at":"2026-08-20T16:31:31.767996Z","submitted_at":"2026-05-08T13:25:27Z","title":"Memory-Efficient Looped Transformer: Decoupling Compute from Memory in Looped Language Models","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-05-11T02:44:36.697450Z"},"links":{"cited_paper":"/paper/2405.12981","citing_paper":"/paper/2605.07721"},"observation_digest":"sha256:1e2950d2cb3de8882fe05408f8af4c07d592a6bc6b3014b547e9fda7b7c0a831","observation_id":"75f6d43b-16fc-4490-bdf2-28c20b0612d6","resolution":{"observed_at":"2026-05-11T02:45:57.887306Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.12981","last_updated":"2024-05-21T17:59:29Z","snapshot_observed_at":"2026-08-20T10:03:23.466960Z","submitted_at":"2024-05-21T17:59:29Z","title":"Reducing Transformer Key-Value Cache Size with Cross-Layer Attention","version":1},"cited_work":{"arxiv_id":"2405.12981","doi":"10.48550/arxiv.2405.12981","metadata_source":"pith","pith_arxiv_id":"2405.12981","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Reducing transformer key-value cache size with cross-layer attention","venue":"cs.LG","work_id":"e92a5a41-6fe4-4a0f-84a0-b866c6b353f3","year":2024},"citing_paper":{"arxiv_id":"2605.07721","last_updated":"2026-05-19T08:08:32Z","snapshot_observed_at":"2026-08-20T16:31:31.767996Z","submitted_at":"2026-05-08T13:25:27Z","title":"Memory-Efficient Looped Transformer: Decoupling Compute from Memory in Looped Language Models","version":2},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-05-20T23:01:17.957634Z"},"links":{"cited_paper":"/paper/2405.12981","citing_paper":"/paper/2605.07721"},"observation_digest":"sha256:ac9e223824a6864f3457b1213058012729c115bd4b49ee66789182946924485d","observation_id":"3673d03d-8509-48e2-be5b-673b600ca091","resolution":{"observed_at":"2026-05-20T23:03:50.629580Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.12981","last_updated":"2024-05-21T17:59:29Z","snapshot_observed_at":"2026-08-20T10:03:23.466960Z","submitted_at":"2024-05-21T17:59:29Z","title":"Reducing Transformer Key-Value Cache Size with Cross-Layer Attention","version":1},"cited_work":{"arxiv_id":"2405.12981","doi":"10.48550/arxiv.2405.12981","metadata_source":"pith","pith_arxiv_id":"2405.12981","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Reducing transformer key-value cache size with cross-layer attention","venue":"cs.LG","work_id":"e92a5a41-6fe4-4a0f-84a0-b866c6b353f3","year":2024},"citing_paper":{"arxiv_id":"2606.02780","last_updated":"2026-08-16T03:22:25Z","snapshot_observed_at":"2026-08-20T23:12:32.926034Z","submitted_at":"2026-06-01T18:43:34Z","title":"Do Value Vectors in Deep Layers Need Context from the Residual Stream?","version":2},"reference_index":64,"source":"arxiv_source","source_observed_at":"2026-06-28T14:35:48.292081Z"},"links":{"cited_paper":"/paper/2405.12981","citing_paper":"/paper/2606.02780"},"observation_digest":"sha256:cf9a04fe8c37636ad7aac230d2473526cc7fc011692dd6fe2cee5894ab484624","observation_id":"3a38211c-62ab-4de3-a429-0a1f48999f38","resolution":{"observed_at":"2026-07-01T23:16:23.469169Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.12981","last_updated":"2024-05-21T17:59:29Z","snapshot_observed_at":"2026-08-20T10:03:23.466960Z","submitted_at":"2024-05-21T17:59:29Z","title":"Reducing Transformer Key-Value Cache Size with Cross-Layer Attention","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.12981","snapshot_observed_at":"2026-08-02T12:41:19.361685Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2606.02780","last_updated":"2026-08-16T03:22:25Z","snapshot_observed_at":"2026-08-20T23:12:32.926034Z","submitted_at":"2026-06-01T18:43:34Z","title":"Do Value Vectors in Deep Layers Need Context from the Residual Stream?","version":3},"reference_index":4,"source":"arxiv_source","source_observed_at":"2026-08-02T12:41:19.361685Z"},"links":{"cited_paper":"/paper/2405.12981","citing_paper":"/paper/2606.02780"},"observation_digest":"sha256:62ed56d14071c5f432079b7c1833c49e4ca06fa814c4d3a234310f67e2277b8e","observation_id":"306faa8c-84e0-49b9-808f-58f778b2b301","resolution":{"observed_at":"2026-08-02T12:41:19.361685Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.12981","last_updated":"2024-05-21T17:59:29Z","snapshot_observed_at":"2026-08-20T10:03:23.466960Z","submitted_at":"2024-05-21T17:59:29Z","title":"Reducing Transformer Key-Value Cache Size with Cross-Layer Attention","version":1},"cited_work":{"arxiv_id":"2405.12981","doi":"10.48550/arxiv.2405.12981","metadata_source":"pith","pith_arxiv_id":"2405.12981","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Reducing transformer key-value cache size with cross-layer attention","venue":"cs.LG","work_id":"e92a5a41-6fe4-4a0f-84a0-b866c6b353f3","year":2024},"citing_paper":{"arxiv_id":"2607.08032","last_updated":"2026-07-09T01:15:03Z","snapshot_observed_at":"2026-08-18T23:35:57.173534Z","submitted_at":"2026-07-09T01:15:03Z","title":"What to Keep, What to Forget: A Rate--Distortion View of Memory Compaction in LLMs and Agents","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-07-10T01:26:59.421158Z"},"links":{"cited_paper":"/paper/2405.12981","citing_paper":"/paper/2607.08032"},"observation_digest":"sha256:271e026e15a1cb9cce3370989988f6b2af74f3e2ec0d794d10e22688fcbfd708","observation_id":"94bdc96c-9f3f-447c-be23-74ce6c3bbf63","resolution":{"observed_at":"2026-07-10T01:36:44.309590Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-21T06:32:19.484+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.12981","last_updated":"2024-05-21T17:59:29Z","snapshot_observed_at":"2026-08-20T10:03:23.466960Z","submitted_at":"2024-05-21T17:59:29Z","title":"Reducing Transformer Key-Value Cache Size with Cross-Layer Attention","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.12981","snapshot_observed_at":"2026-08-02T09:51:03.342951Z","title":"arXiv preprint arXiv:2405.12981 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.16252","last_updated":"2026-06-27T03:25:36Z","snapshot_observed_at":"2026-08-16T23:14:00.224337Z","submitted_at":"2026-06-27T03:25:36Z","title":"SOS-LoRA: Static Orthogonal-Subspace Low-Rank Adaptation with Fixed Multi-Scale Scaling","version":1},"reference_index":28,"source":"arxiv_source","source_observed_at":"2026-08-02T09:51:03.342951Z"},"links":{"cited_paper":"/paper/2405.12981","citing_paper":"/paper/2607.16252"},"observation_digest":"sha256:a5a7a8a6edb8f75fc916b255f781613496279db85f197e8b8d7e83122a70011e","observation_id":"347be885-1d25-4ac8-91f9-9c10bb0815f5","resolution":{"observed_at":"2026-08-02T09:51:03.342951Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2405.12981/citation-record","integrity":"/paper/2405.12981/integrity","json":"/paper/2405.12981/citation-record.json","paper":"/paper/2405.12981"},"outbound":[],"paper":{"arxiv_id":"2405.12981","last_updated":"2024-05-21T17:59:29Z","latest_version":1,"primary_category":"cs.LG","snapshot_observed_at":"2026-08-20T10:03:23.466960Z","submitted_at":"2024-05-21T17:59:29Z","title":"Reducing Transformer Key-Value Cache Size with Cross-Layer Attention"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-21T06:32:19.484+00:00","source":"crossref"},{"observed_at":"2026-08-21T06:32:16.066871+00:00","source":"retraction_watch"}],"thesis":"As of 21 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 30 inbound Pith citation observations for arXiv:2405.12981."}