{"as_of":"2026-08-05T07:54:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:082c859ce994a58a87f4147271ff1771ad41e700b677fa831b8f5133f8f8d890","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":16,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":16,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-05T06:32:48.257954+00:00","state":"measured"},{"denominator":16,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":16,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-04T20:53:30.115688Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":1,"source":"arxiv_reference","source_observed_at":"2026-08-05T02:28:24.338817Z","state":"measured"}],"external_citation_measurements":[{"count":2,"observed_at":"2026-08-05T02:28:24.338817Z","source":"arxiv_reference"}],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2407.18003","last_updated":"2024-11-20T02:04:10Z","snapshot_observed_at":"2026-07-06T18:51:49.871551Z","submitted_at":"2024-07-25T12:56:22Z","title":"Keep the Cost Down: A Review on Methods to Optimize LLM' s KV-Cache Consumption","version":4},"cited_work":{"arxiv_id":"2407.18003","doi":"10.48550/arxiv.2407.18003","metadata_source":"arxiv_reference","pith_arxiv_id":"2407.18003","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Keep the cost down: A review on methods to optimize llm’s kv-cache consumption","venue":"arXiv (Cornell University)","work_id":"45195b7c-36a4-4e3e-8f9e-db76070aec96","year":2024},"citing_paper":{"arxiv_id":"2410.13846","last_updated":"2026-05-18T05:12:14Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-10-17T17:58:14Z","title":"LightTransfer: Your Long-Context LLM is Secretly a Hybrid Model with Effortless Adaptation","version":3},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-05-23T18:31:35.391674Z"},"links":{"cited_paper":"/paper/2407.18003","citing_paper":"/paper/2410.13846"},"observation_digest":"sha256:ccfe5c8fabbe002437af69692ce8eeb24b2fd0a55d0d81c5728d572e076690a7","observation_id":"5219d9de-e731-44a0-90d5-02d903ff308e","resolution":{"observed_at":"2026-05-23T18:33:19.467156Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.18003","last_updated":"2024-11-20T02:04:10Z","snapshot_observed_at":"2026-07-06T18:51:49.871551Z","submitted_at":"2024-07-25T12:56:22Z","title":"Keep the Cost Down: A Review on Methods to Optimize LLM' s KV-Cache Consumption","version":4},"cited_work":{"arxiv_id":"2407.18003","doi":"10.48550/arxiv.2407.18003","metadata_source":"arxiv_reference","pith_arxiv_id":"2407.18003","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Keep the cost down: A review on methods to optimize llm’s kv-cache consumption","venue":"arXiv (Cornell University)","work_id":"45195b7c-36a4-4e3e-8f9e-db76070aec96","year":2024},"citing_paper":{"arxiv_id":"2502.20295","last_updated":"2026-04-18T08:27:37Z","snapshot_observed_at":"2026-08-02T21:22:35.616248Z","submitted_at":"2025-02-27T17:21:18Z","title":"Judge a Book by its Cover: Investigating Multi-Modal LLMs for Multi-Page Handwritten Document Transcription","version":2},"reference_index":38,"source":"arxiv_source","source_observed_at":"2026-05-23T02:23:22.682357Z"},"links":{"cited_paper":"/paper/2407.18003","citing_paper":"/paper/2502.20295"},"observation_digest":"sha256:6731e93602f6d6c8f63a26aa719b8b4602edcd73550565cb65801efd4bafd20c","observation_id":"d9d6c010-c1f2-4d1a-ad8e-4885cbdcba68","resolution":{"observed_at":"2026-05-23T02:25:19.420768Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.18003","last_updated":"2024-11-20T02:04:10Z","snapshot_observed_at":"2026-07-06T18:51:49.871551Z","submitted_at":"2024-07-25T12:56:22Z","title":"Keep the Cost Down: A Review on Methods to Optimize LLM' s KV-Cache Consumption","version":4},"cited_work":{"arxiv_id":"2407.18003","doi":"10.48550/arxiv.2407.18003","metadata_source":"arxiv_reference","pith_arxiv_id":"2407.18003","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Keep the cost down: A review on methods to optimize llm’s kv-cache consumption","venue":"arXiv (Cornell University)","work_id":"45195b7c-36a4-4e3e-8f9e-db76070aec96","year":2024},"citing_paper":{"arxiv_id":"2503.16419","last_updated":"2025-08-21T19:14:40Z","snapshot_observed_at":"2026-08-03T03:23:56.857266Z","submitted_at":"2025-03-20T17:59:38Z","title":"Stop Overthinking: A Survey on Efficient Reasoning for Large Language Models","version":4},"reference_index":156,"source":"pdf_text","source_observed_at":"2026-05-14T01:29:56.480020Z"},"links":{"cited_paper":"/paper/2407.18003","citing_paper":"/paper/2503.16419"},"observation_digest":"sha256:e1804965627890ee6961be54a35ef342689c3f5b2e43aae624e08ebee2c04fdf","observation_id":"a6650d7d-366e-4a5c-96bd-68c48e7e7dc7","resolution":{"observed_at":"2026-05-14T01:29:56.798694Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.18003","last_updated":"2024-11-20T02:04:10Z","snapshot_observed_at":"2026-07-06T18:51:49.871551Z","submitted_at":"2024-07-25T12:56:22Z","title":"Keep the Cost Down: A Review on Methods to Optimize LLM' s KV-Cache Consumption","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.18003","snapshot_observed_at":"2026-08-04T20:53:30.115688Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.08315","last_updated":"2025-09-10T06:32:49Z","snapshot_observed_at":"2026-08-04T20:53:17.807634Z","submitted_at":"2025-09-10T06:32:49Z","title":"EvolKV: Evolutionary KV Cache Compression for LLM Inference","version":1},"reference_index":42,"source":"arxiv_source","source_observed_at":"2026-08-04T20:53:30.115688Z"},"links":{"cited_paper":"/paper/2407.18003","citing_paper":"/paper/2509.08315"},"observation_digest":"sha256:92ff861ff4a9c2776c73177647ce379315342835ca38a17ddba49dd591484a35","observation_id":"64453fe3-4c28-4191-8b70-bc5a7136d4cb","resolution":{"observed_at":"2026-08-04T20:53:30.115688Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.18003","last_updated":"2024-11-20T02:04:10Z","snapshot_observed_at":"2026-07-06T18:51:49.871551Z","submitted_at":"2024-07-25T12:56:22Z","title":"Keep the Cost Down: A Review on Methods to Optimize LLM' s KV-Cache Consumption","version":4},"cited_work":{"arxiv_id":"2407.18003","doi":"10.48550/arxiv.2407.18003","metadata_source":"arxiv_reference","pith_arxiv_id":"2407.18003","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Keep the cost down: A review on methods to optimize llm’s kv-cache consumption","venue":"arXiv (Cornell University)","work_id":"45195b7c-36a4-4e3e-8f9e-db76070aec96","year":2024},"citing_paper":{"arxiv_id":"2509.21623","last_updated":"2026-04-16T21:29:54Z","snapshot_observed_at":"2026-08-03T21:57:13.209890Z","submitted_at":"2025-09-25T21:42:27Z","title":"OjaKV: Context-Aware Online Low-Rank KV Cache Compression","version":2},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-05-18T13:26:02.980973Z"},"links":{"cited_paper":"/paper/2407.18003","citing_paper":"/paper/2509.21623"},"observation_digest":"sha256:e30779f0b2c1c100ba49da38f9d7c78213a15f7e130478045d636b01f1dcc4de","observation_id":"2eb1b009-0208-437c-85a3-eaf280a9a3db","resolution":{"observed_at":"2026-05-18T13:26:24.702813Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.18003","last_updated":"2024-11-20T02:04:10Z","snapshot_observed_at":"2026-07-06T18:51:49.871551Z","submitted_at":"2024-07-25T12:56:22Z","title":"Keep the Cost Down: A Review on Methods to Optimize LLM' s KV-Cache Consumption","version":4},"cited_work":{"arxiv_id":"2407.18003","doi":"10.48550/arxiv.2407.18003","metadata_source":"arxiv_reference","pith_arxiv_id":"2407.18003","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Keep the cost down: A review on methods to optimize llm’s kv-cache consumption","venue":"arXiv (Cornell University)","work_id":"45195b7c-36a4-4e3e-8f9e-db76070aec96","year":2024},"citing_paper":{"arxiv_id":"2602.10718","last_updated":"2026-04-28T04:57:29Z","snapshot_observed_at":"2026-07-06T22:45:25.047172Z","submitted_at":"2026-02-11T10:24:42Z","title":"SnapMLA: Efficient Long-Context MLA Decoding via Hardware-Aware FP8 Quantized Pipelining","version":3},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-05-16T05:58:03.113220Z"},"links":{"cited_paper":"/paper/2407.18003","citing_paper":"/paper/2602.10718"},"observation_digest":"sha256:d2009683bf333e4b3c9ea13a7b14d529fd4aa5ff3ff06c50f76d97ea727dd650","observation_id":"b21af5fd-a71a-4e57-a297-453a0d1f8cd8","resolution":{"observed_at":"2026-05-16T06:00:40.765255Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.18003","last_updated":"2024-11-20T02:04:10Z","snapshot_observed_at":"2026-07-06T18:51:49.871551Z","submitted_at":"2024-07-25T12:56:22Z","title":"Keep the Cost Down: A Review on Methods to Optimize LLM' s KV-Cache Consumption","version":4},"cited_work":{"arxiv_id":"2407.18003","doi":"10.48550/arxiv.2407.18003","metadata_source":"arxiv_reference","pith_arxiv_id":"2407.18003","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Keep the cost down: A review on methods to optimize llm’s kv-cache consumption","venue":"arXiv (Cornell University)","work_id":"45195b7c-36a4-4e3e-8f9e-db76070aec96","year":2024},"citing_paper":{"arxiv_id":"2603.22910","last_updated":"2026-05-13T05:25:05Z","snapshot_observed_at":"2026-07-06T22:50:14.856451Z","submitted_at":"2026-03-24T07:58:42Z","title":"EchoKV: Efficient KV Cache Compression via Similarity-Based Reconstruction","version":2},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-05-15T01:09:07.983785Z"},"links":{"cited_paper":"/paper/2407.18003","citing_paper":"/paper/2603.22910"},"observation_digest":"sha256:e6c43745271237e9c503aff7e549fa59838f85d11f6fa5907b2e3dffd7119af6","observation_id":"ef827935-40c0-4108-87f3-cef628ce9116","resolution":{"observed_at":"2026-05-15T01:09:36.990898Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.18003","last_updated":"2024-11-20T02:04:10Z","snapshot_observed_at":"2026-07-06T18:51:49.871551Z","submitted_at":"2024-07-25T12:56:22Z","title":"Keep the Cost Down: A Review on Methods to Optimize LLM' s KV-Cache Consumption","version":4},"cited_work":{"arxiv_id":"2407.18003","doi":"10.48550/arxiv.2407.18003","metadata_source":"arxiv_reference","pith_arxiv_id":"2407.18003","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Keep the cost down: A review on methods to optimize llm’s kv-cache consumption","venue":"arXiv (Cornell University)","work_id":"45195b7c-36a4-4e3e-8f9e-db76070aec96","year":2024},"citing_paper":{"arxiv_id":"2604.04500","last_updated":"2026-04-06T07:51:59Z","snapshot_observed_at":"2026-07-30T06:32:02.248567Z","submitted_at":"2026-04-06T07:51:59Z","title":"Saliency-R1: Enforcing Interpretable and Faithful Vision-language Reasoning via Saliency-map Alignment Reward","version":1},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-05-10T19:59:19.379119Z"},"links":{"cited_paper":"/paper/2407.18003","citing_paper":"/paper/2604.04500"},"observation_digest":"sha256:6ced6571b056d8a4f35fa62fcfa135046af2ccb858167979296d4572a1472046","observation_id":"f78f4f86-0f53-4092-a17d-fcd360ec2729","resolution":{"observed_at":"2026-05-10T22:20:47.770772Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.18003","last_updated":"2024-11-20T02:04:10Z","snapshot_observed_at":"2026-07-06T18:51:49.871551Z","submitted_at":"2024-07-25T12:56:22Z","title":"Keep the Cost Down: A Review on Methods to Optimize LLM' s KV-Cache Consumption","version":4},"cited_work":{"arxiv_id":"2407.18003","doi":"10.48550/arxiv.2407.18003","metadata_source":"arxiv_reference","pith_arxiv_id":"2407.18003","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Keep the cost down: A review on methods to optimize llm’s kv-cache consumption","venue":"arXiv (Cornell University)","work_id":"45195b7c-36a4-4e3e-8f9e-db76070aec96","year":2024},"citing_paper":{"arxiv_id":"2604.04921","last_updated":"2026-04-06T17:58:42Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2026-04-06T17:58:42Z","title":"TriAttention: Efficient Long Reasoning with Trigonometric KV Compression","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-05-10T19:55:45.739280Z"},"links":{"cited_paper":"/paper/2407.18003","citing_paper":"/paper/2604.04921"},"observation_digest":"sha256:ef01c51dc04ae49f5d5075a34be373f9bcbadcab891728b89a1fc07b8f28340e","observation_id":"842f3199-2da6-41b7-be1e-3e9db7153cfb","resolution":{"observed_at":"2026-05-10T22:20:49.203487Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.18003","last_updated":"2024-11-20T02:04:10Z","snapshot_observed_at":"2026-07-06T18:51:49.871551Z","submitted_at":"2024-07-25T12:56:22Z","title":"Keep the Cost Down: A Review on Methods to Optimize LLM' s KV-Cache Consumption","version":4},"cited_work":{"arxiv_id":"2407.18003","doi":"10.48550/arxiv.2407.18003","metadata_source":"arxiv_reference","pith_arxiv_id":"2407.18003","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Keep the cost down: A review on methods to optimize llm’s kv-cache consumption","venue":"arXiv (Cornell University)","work_id":"45195b7c-36a4-4e3e-8f9e-db76070aec96","year":2024},"citing_paper":{"arxiv_id":"2604.17935","last_updated":"2026-04-20T08:15:17Z","snapshot_observed_at":"2026-07-06T23:04:55.190916Z","submitted_at":"2026-04-20T08:15:17Z","title":"How Much Cache Does Reasoning Need? Depth-Cache Tradeoffs in KV-Compressed Transformers","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-05-10T05:17:52.344313Z"},"links":{"cited_paper":"/paper/2407.18003","citing_paper":"/paper/2604.17935"},"observation_digest":"sha256:80ead2bf95c16c7c60b9dba8b1f2ee443523b4feba512851f308fc02e320b452","observation_id":"5da7ee58-3e80-436a-8159-41aac99f4f6c","resolution":{"observed_at":"2026-05-10T09:28:39.191396Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.18003","last_updated":"2024-11-20T02:04:10Z","snapshot_observed_at":"2026-07-06T18:51:49.871551Z","submitted_at":"2024-07-25T12:56:22Z","title":"Keep the Cost Down: A Review on Methods to Optimize LLM' s KV-Cache Consumption","version":4},"cited_work":{"arxiv_id":"2407.18003","doi":"10.48550/arxiv.2407.18003","metadata_source":"arxiv_reference","pith_arxiv_id":"2407.18003","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Keep the cost down: A review on methods to optimize llm’s kv-cache consumption","venue":"arXiv (Cornell University)","work_id":"45195b7c-36a4-4e3e-8f9e-db76070aec96","year":2024},"citing_paper":{"arxiv_id":"2604.25975","last_updated":"2026-04-28T12:28:04Z","snapshot_observed_at":"2026-07-06T23:11:43.764239Z","submitted_at":"2026-04-28T12:28:04Z","title":"Rethinking KV Cache Eviction via a Unified Information-Theoretic Objective","version":1},"reference_index":26,"source":"arxiv_source","source_observed_at":"2026-05-07T16:41:23.234607Z"},"links":{"cited_paper":"/paper/2407.18003","citing_paper":"/paper/2604.25975"},"observation_digest":"sha256:479f0b081abfa9ca9f5b34a33e6c806352181000b2704aeb59ce0f59f0e960a7","observation_id":"8dce6e59-b5f3-4860-ba32-eb1112d25229","resolution":{"observed_at":"2026-05-09T01:54:34.032488Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.18003","last_updated":"2024-11-20T02:04:10Z","snapshot_observed_at":"2026-07-06T18:51:49.871551Z","submitted_at":"2024-07-25T12:56:22Z","title":"Keep the Cost Down: A Review on Methods to Optimize LLM' s KV-Cache Consumption","version":4},"cited_work":{"arxiv_id":"2407.18003","doi":"10.48550/arxiv.2407.18003","metadata_source":"arxiv_reference","pith_arxiv_id":"2407.18003","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Keep the cost down: A review on methods to optimize llm’s kv-cache consumption","venue":"arXiv (Cornell University)","work_id":"45195b7c-36a4-4e3e-8f9e-db76070aec96","year":2024},"citing_paper":{"arxiv_id":"2605.04901","last_updated":"2026-05-06T13:31:15Z","snapshot_observed_at":"2026-07-06T23:17:38.226365Z","submitted_at":"2026-05-06T13:31:15Z","title":"On the (In-)Security of the Shuffling Defense in the Transformer Secure Inference","version":1},"reference_index":9,"source":"arxiv_source","source_observed_at":"2026-05-08T17:24:04.123827Z"},"links":{"cited_paper":"/paper/2407.18003","citing_paper":"/paper/2605.04901"},"observation_digest":"sha256:12851467e2f280bff7b97d14c903fdce46d177a11b739b908f3fbc61a739f1b6","observation_id":"2b58e1d1-177a-4766-91d7-f114ac55f69b","resolution":{"observed_at":"2026-05-11T17:36:06.555949Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.18003","last_updated":"2024-11-20T02:04:10Z","snapshot_observed_at":"2026-07-06T18:51:49.871551Z","submitted_at":"2024-07-25T12:56:22Z","title":"Keep the Cost Down: A Review on Methods to Optimize LLM' s KV-Cache Consumption","version":4},"cited_work":{"arxiv_id":"2407.18003","doi":"10.48550/arxiv.2407.18003","metadata_source":"arxiv_reference","pith_arxiv_id":"2407.18003","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Keep the cost down: A review on methods to optimize llm’s kv-cache consumption","venue":"arXiv (Cornell University)","work_id":"45195b7c-36a4-4e3e-8f9e-db76070aec96","year":2024},"citing_paper":{"arxiv_id":"2605.06165","last_updated":"2026-05-07T12:51:49Z","snapshot_observed_at":"2026-07-06T23:18:41.400741Z","submitted_at":"2026-05-07T12:51:49Z","title":"Post Reasoning: Improving the Performance of Non-Thinking Models at No Cost","version":1},"reference_index":168,"source":"arxiv_source","source_observed_at":"2026-05-08T10:19:08.451445Z"},"links":{"cited_paper":"/paper/2407.18003","citing_paper":"/paper/2605.06165"},"observation_digest":"sha256:09faf48303a458ad41339d175895fb93e5cd3a7564f7f153178af1b060583438","observation_id":"2d31888f-da9b-495c-a3fa-7b03b3942266","resolution":{"observed_at":"2026-05-11T20:06:09.876717Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.18003","last_updated":"2024-11-20T02:04:10Z","snapshot_observed_at":"2026-07-06T18:51:49.871551Z","submitted_at":"2024-07-25T12:56:22Z","title":"Keep the Cost Down: A Review on Methods to Optimize LLM' s KV-Cache Consumption","version":4},"cited_work":{"arxiv_id":"2407.18003","doi":"10.48550/arxiv.2407.18003","metadata_source":"arxiv_reference","pith_arxiv_id":"2407.18003","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Keep the cost down: A review on methods to optimize llm’s kv-cache consumption","venue":"arXiv (Cornell University)","work_id":"45195b7c-36a4-4e3e-8f9e-db76070aec96","year":2024},"citing_paper":{"arxiv_id":"2605.11516","last_updated":"2026-05-12T04:39:34Z","snapshot_observed_at":"2026-07-06T23:23:21.034839Z","submitted_at":"2026-05-12T04:39:34Z","title":"Agents Should Replace Narrow Predictive AI as the Orchestrator in 6G AI-RAN","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-05-13T02:12:21.409833Z"},"links":{"cited_paper":"/paper/2407.18003","citing_paper":"/paper/2605.11516"},"observation_digest":"sha256:b9acc37e0fdbf608b43d134615d71d1f9d385d7227e66d2bddb0fa6da34620e8","observation_id":"7804a3b1-2b9e-4312-924b-2cf77aa08c07","resolution":{"observed_at":"2026-05-13T02:17:06.969873Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.18003","last_updated":"2024-11-20T02:04:10Z","snapshot_observed_at":"2026-07-06T18:51:49.871551Z","submitted_at":"2024-07-25T12:56:22Z","title":"Keep the Cost Down: A Review on Methods to Optimize LLM' s KV-Cache Consumption","version":4},"cited_work":{"arxiv_id":"2407.18003","doi":"10.48550/arxiv.2407.18003","metadata_source":"arxiv_reference","pith_arxiv_id":"2407.18003","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Keep the cost down: A review on methods to optimize llm’s kv-cache consumption","venue":"arXiv (Cornell University)","work_id":"45195b7c-36a4-4e3e-8f9e-db76070aec96","year":2024},"citing_paper":{"arxiv_id":"2606.00620","last_updated":"2026-05-30T08:51:28Z","snapshot_observed_at":"2026-07-06T23:41:19.955034Z","submitted_at":"2026-05-30T08:51:28Z","title":"FlowNar: Scalable Streaming Narration for Long-Form Videos","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-06-28T19:08:36.655886Z"},"links":{"cited_paper":"/paper/2407.18003","citing_paper":"/paper/2606.00620"},"observation_digest":"sha256:348e4d2fd88ef98c8324e1241547248198b5753171b5cc380c3970cc2ded38b4","observation_id":"444c5df5-e6c5-478e-b03c-62bc57f60cfa","resolution":{"observed_at":"2026-06-28T19:12:34.742615Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.18003","last_updated":"2024-11-20T02:04:10Z","snapshot_observed_at":"2026-07-06T18:51:49.871551Z","submitted_at":"2024-07-25T12:56:22Z","title":"Keep the Cost Down: A Review on Methods to Optimize LLM' s KV-Cache Consumption","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.18003","snapshot_observed_at":"2026-07-14T06:16:09.070414Z","title":"arXiv preprint arXiv:2407.18003 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.11211","last_updated":"2026-07-13T08:02:26Z","snapshot_observed_at":"2026-07-16T23:19:15.811399Z","submitted_at":"2026-07-13T08:02:26Z","title":"FastTPS: An Optimized Method for LLM Token Phase for AI accelerators","version":1},"reference_index":40,"source":"arxiv_source","source_observed_at":"2026-07-14T06:16:09.070414Z"},"links":{"cited_paper":"/paper/2407.18003","citing_paper":"/paper/2607.11211"},"observation_digest":"sha256:8eb7e43f78b3727fd545b99a66bfc33ea514b520db761742602e49d2056a7054","observation_id":"95bdb966-0c58-4681-837b-d792a80376a8","resolution":{"observed_at":"2026-07-14T06:16:09.070414Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2407.18003/citation-record","integrity":"/paper/2407.18003/integrity","json":"/paper/2407.18003/citation-record.json","paper":"/paper/2407.18003"},"outbound":[],"paper":{"arxiv_id":"2407.18003","last_updated":"2024-11-20T02:04:10Z","latest_version":4,"primary_category":"cs.CL","snapshot_observed_at":"2026-07-06T18:51:49.871551Z","submitted_at":"2024-07-25T12:56:22Z","title":"Keep the Cost Down: A Review on Methods to Optimize LLM' s KV-Cache Consumption"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"thesis":"As of 5 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 16 inbound Pith citation observations for arXiv:2407.18003."}