{"as_of":"2026-08-06T16:26:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:2237bec5dd80316872e5c93d3d0ea5cc74589bfbb5f6fd3d5a4667794b169680","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":14,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":14,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-06T06:34:29.942622+00:00","state":"measured"},{"denominator":14,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":14,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-06T10:22:42.522976Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":1,"source":"pith","source_observed_at":"2026-08-05T02:28:24.338817Z","state":"measured"}],"external_citation_measurements":[{"count":1,"observed_at":"2026-08-05T02:28:24.338817Z","source":"pith"}],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2403.19708","last_updated":"2024-06-30T23:50:38Z","snapshot_observed_at":"2026-08-03T11:15:07.121206Z","submitted_at":"2024-03-23T10:42:49Z","title":"Cost-Efficient Large Language Model Serving for Multi-turn Conversations with CachedAttention","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.19708","snapshot_observed_at":"2026-08-06T10:22:42.522976Z","title":"AttentionStore: Cost-effective attention reuse across multi- turn conversations in large language model serving,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.00234","last_updated":"2025-08-01T00:45:15Z","snapshot_observed_at":"2026-08-06T10:22:40.273970Z","submitted_at":"2025-08-01T00:45:15Z","title":"Quality-of-Service Aware LLM Routing for Edge Computing with Multiple Experts","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-06T10:22:42.522976Z"},"links":{"cited_paper":"/paper/2403.19708","citing_paper":"/paper/2508.00234"},"observation_digest":"sha256:85f13eaf02ffdbc9dc611f2d096ce2a9d44f16d4ad7ff54e5fb7a805a3667f59","observation_id":"120312c5-8335-48a5-8248-6cb87de67a0f","resolution":{"observed_at":"2026-08-06T10:22:42.522976Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.19708","last_updated":"2024-06-30T23:50:38Z","snapshot_observed_at":"2026-08-03T11:15:07.121206Z","submitted_at":"2024-03-23T10:42:49Z","title":"Cost-Efficient Large Language Model Serving for Multi-turn Conversations with CachedAttention","version":3},"cited_work":{"arxiv_id":"2403.19708","doi":"10.48550/arxiv.2403.19708","metadata_source":"pith","pith_arxiv_id":"2403.19708","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Attentionstore: Cost-effective atten- tion reuse across multi-turn conversations in large lan- guage model serving.arXiv preprint arXiv:2403.19708, 52:20–38","venue":"cs.CL","work_id":"d40a491e-677b-4dff-a6cb-636a8034acbc","year":2024},"citing_paper":{"arxiv_id":"2511.02230","last_updated":"2026-05-25T23:34:23Z","snapshot_observed_at":"2026-08-04T00:19:14.332924Z","submitted_at":"2025-11-04T03:43:05Z","title":"Continuum: Efficient and Robust Multi-Turn LLM Agent Scheduling with KV Cache Time-to-Live","version":5},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-05-18T01:58:23.234348Z"},"links":{"cited_paper":"/paper/2403.19708","citing_paper":"/paper/2511.02230"},"observation_digest":"sha256:88f65b585cac518446c317941975fb518e81171252ad0c2f8db03587ed49d93d","observation_id":"6da3e595-ad2f-45d1-9bb1-b5038c8a11c0","resolution":{"observed_at":"2026-05-18T02:00:39.752838Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.19708","last_updated":"2024-06-30T23:50:38Z","snapshot_observed_at":"2026-08-03T11:15:07.121206Z","submitted_at":"2024-03-23T10:42:49Z","title":"Cost-Efficient Large Language Model Serving for Multi-turn Conversations with CachedAttention","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.19708","snapshot_observed_at":"2026-08-04T00:19:17.443187Z","title":"Attentionstore: Cost-effective atten- tion reuse across multi-turn conversations in large lan- guage model serving.arXiv preprint arXiv:2403.19708, 52:20–38, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2511.02230","last_updated":"2026-05-25T23:34:23Z","snapshot_observed_at":"2026-08-04T00:19:14.332924Z","submitted_at":"2025-11-04T03:43:05Z","title":"Continuum: Efficient and Robust Multi-Turn LLM Agent Scheduling with KV Cache Time-to-Live","version":6},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-04T00:19:17.443187Z"},"links":{"cited_paper":"/paper/2403.19708","citing_paper":"/paper/2511.02230"},"observation_digest":"sha256:86b44d294a1919be1528a6cff67978ab5bec9cffd24762134949a030bbf74d68","observation_id":"b8e1e9e6-cf7e-48cc-af69-590c16514dfa","resolution":{"observed_at":"2026-08-04T00:19:17.443187Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.19708","last_updated":"2024-06-30T23:50:38Z","snapshot_observed_at":"2026-08-03T11:15:07.121206Z","submitted_at":"2024-03-23T10:42:49Z","title":"Cost-Efficient Large Language Model Serving for Multi-turn Conversations with CachedAttention","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.19708","snapshot_observed_at":"2026-07-13T13:32:41.086180Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2604.03137","last_updated":"2026-06-16T15:46:53Z","snapshot_observed_at":"2026-07-13T13:32:40.187094Z","submitted_at":"2026-04-03T15:57:32Z","title":"Low-Scaling Many-Body Green's Function Calculations for Molecular Systems via Interacting-Bath Dynamical Embedding Theory","version":2},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-07-13T13:32:41.086180Z"},"links":{"cited_paper":"/paper/2403.19708","citing_paper":"/paper/2604.03137"},"observation_digest":"sha256:d5a8777a90a011c17e2d7014c69e7c5606f35111788fcc31ae0825c0925a9c53","observation_id":"c62d904d-4177-4a83-87e2-f7e3a47616ac","resolution":{"observed_at":"2026-07-13T13:32:41.086180Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.19708","last_updated":"2024-06-30T23:50:38Z","snapshot_observed_at":"2026-08-03T11:15:07.121206Z","submitted_at":"2024-03-23T10:42:49Z","title":"Cost-Efficient Large Language Model Serving for Multi-turn Conversations with CachedAttention","version":3},"cited_work":{"arxiv_id":"2403.19708","doi":"10.48550/arxiv.2403.19708","metadata_source":"pith","pith_arxiv_id":"2403.19708","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Attentionstore: Cost-effective atten- tion reuse across multi-turn conversations in large lan- guage model serving.arXiv preprint arXiv:2403.19708, 52:20–38","venue":"cs.CL","work_id":"d40a491e-677b-4dff-a6cb-636a8034acbc","year":2024},"citing_paper":{"arxiv_id":"2604.03143","last_updated":"2026-04-03T16:04:40Z","snapshot_observed_at":"2026-07-06T22:52:25.307426Z","submitted_at":"2026-04-03T16:04:40Z","title":"TokenDance: Scaling Multi-Agent LLM Serving via Collective KV Cache Sharing","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-05-13T18:02:46.266111Z"},"links":{"cited_paper":"/paper/2403.19708","citing_paper":"/paper/2604.03143"},"observation_digest":"sha256:e451174b6038870f6a95808535169e3a451722655271723057f1a2b0a10a3757","observation_id":"2ded0a11-a249-4292-b383-e932afa77729","resolution":{"observed_at":"2026-05-13T18:03:04.268884Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.19708","last_updated":"2024-06-30T23:50:38Z","snapshot_observed_at":"2026-08-03T11:15:07.121206Z","submitted_at":"2024-03-23T10:42:49Z","title":"Cost-Efficient Large Language Model Serving for Multi-turn Conversations with CachedAttention","version":3},"cited_work":{"arxiv_id":"2403.19708","doi":"10.48550/arxiv.2403.19708","metadata_source":"pith","pith_arxiv_id":"2403.19708","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Attentionstore: Cost-effective atten- tion reuse across multi-turn conversations in large lan- guage model serving.arXiv preprint arXiv:2403.19708, 52:20–38","venue":"cs.CL","work_id":"d40a491e-677b-4dff-a6cb-636a8034acbc","year":2024},"citing_paper":{"arxiv_id":"2604.17353","last_updated":"2026-04-19T09:59:35Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2026-04-19T09:59:35Z","title":"Hive: A Multi-Agent Infrastructure for Algorithm- and Task-Level Scaling","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-05-10T06:31:36.776819Z"},"links":{"cited_paper":"/paper/2403.19708","citing_paper":"/paper/2604.17353"},"observation_digest":"sha256:95cb1499d8d5f41535275acef2afa80767aef54aed0f86916b546ad13b5f6f6c","observation_id":"ef6f2ec0-955f-444e-9a95-a4319a2c7b74","resolution":{"observed_at":"2026-05-10T06:36:36.708261Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.19708","last_updated":"2024-06-30T23:50:38Z","snapshot_observed_at":"2026-08-03T11:15:07.121206Z","submitted_at":"2024-03-23T10:42:49Z","title":"Cost-Efficient Large Language Model Serving for Multi-turn Conversations with CachedAttention","version":3},"cited_work":{"arxiv_id":"2403.19708","doi":"10.48550/arxiv.2403.19708","metadata_source":"pith","pith_arxiv_id":"2403.19708","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Attentionstore: Cost-effective atten- tion reuse across multi-turn conversations in large lan- guage model serving.arXiv preprint arXiv:2403.19708, 52:20–38","venue":"cs.CL","work_id":"d40a491e-677b-4dff-a6cb-636a8034acbc","year":2024},"citing_paper":{"arxiv_id":"2605.03375","last_updated":"2026-05-05T05:33:11Z","snapshot_observed_at":"2026-07-06T23:16:20.322265Z","submitted_at":"2026-05-05T05:33:11Z","title":"Tutti: Making SSD-Backed KV Cache Practical for Long-Context LLM Serving","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-05-09T16:19:33.613685Z"},"links":{"cited_paper":"/paper/2403.19708","citing_paper":"/paper/2605.03375"},"observation_digest":"sha256:a993436e30fb96f18a2f7572313ee520a4eb254250e0be88a5eed0664861ffec","observation_id":"ed68c6de-3bca-45cd-aa64-cbabb88a2c73","resolution":{"observed_at":"2026-05-11T16:31:10.503096Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.19708","last_updated":"2024-06-30T23:50:38Z","snapshot_observed_at":"2026-08-03T11:15:07.121206Z","submitted_at":"2024-03-23T10:42:49Z","title":"Cost-Efficient Large Language Model Serving for Multi-turn Conversations with CachedAttention","version":3},"cited_work":{"arxiv_id":"2403.19708","doi":"10.48550/arxiv.2403.19708","metadata_source":"pith","pith_arxiv_id":"2403.19708","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Attentionstore: Cost-effective atten- tion reuse across multi-turn conversations in large lan- guage model serving.arXiv preprint arXiv:2403.19708, 52:20–38","venue":"cs.CL","work_id":"d40a491e-677b-4dff-a6cb-636a8034acbc","year":2024},"citing_paper":{"arxiv_id":"2605.11232","last_updated":"2026-05-11T20:47:41Z","snapshot_observed_at":"2026-07-06T23:23:06.755816Z","submitted_at":"2026-05-11T20:47:41Z","title":"Rethinking LLMOps for Fraud and AML: Building a Compliance-Grade LLM Serving Stack","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-13T02:11:18.259161Z"},"links":{"cited_paper":"/paper/2403.19708","citing_paper":"/paper/2605.11232"},"observation_digest":"sha256:0b30a75d90ef245d88ab1a9bd17d56aab07b587003c25fd8cc5e7c7ce90f6065","observation_id":"a004ef01-6537-40fa-b2a6-994d5438e679","resolution":{"observed_at":"2026-05-13T02:12:07.134044Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.19708","last_updated":"2024-06-30T23:50:38Z","snapshot_observed_at":"2026-08-03T11:15:07.121206Z","submitted_at":"2024-03-23T10:42:49Z","title":"Cost-Efficient Large Language Model Serving for Multi-turn Conversations with CachedAttention","version":3},"cited_work":{"arxiv_id":"2403.19708","doi":"10.48550/arxiv.2403.19708","metadata_source":"pith","pith_arxiv_id":"2403.19708","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Attentionstore: Cost-effective atten- tion reuse across multi-turn conversations in large lan- guage model serving.arXiv preprint arXiv:2403.19708, 52:20–38","venue":"cs.CL","work_id":"d40a491e-677b-4dff-a6cb-636a8034acbc","year":2024},"citing_paper":{"arxiv_id":"2606.02964","last_updated":"2026-06-01T23:51:37Z","snapshot_observed_at":"2026-08-02T06:03:10.694543Z","submitted_at":"2026-06-01T23:51:37Z","title":"Multi-Segment Attention: Enabling Efficient KV-Cache Management for Faster Large Language Model Serving","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-06-28T11:38:05.435505Z"},"links":{"cited_paper":"/paper/2403.19708","citing_paper":"/paper/2606.02964"},"observation_digest":"sha256:8b0e5adde837f60ecd3c7aa271a6b447422597359d262b32ca9cd5628435c484","observation_id":"a3af4b67-9b5d-4ee1-b048-6b7d7a7b58e8","resolution":{"observed_at":"2026-07-02T01:36:26.252091Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.19708","last_updated":"2024-06-30T23:50:38Z","snapshot_observed_at":"2026-08-03T11:15:07.121206Z","submitted_at":"2024-03-23T10:42:49Z","title":"Cost-Efficient Large Language Model Serving for Multi-turn Conversations with CachedAttention","version":3},"cited_work":{"arxiv_id":"2403.19708","doi":"10.48550/arxiv.2403.19708","metadata_source":"pith","pith_arxiv_id":"2403.19708","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Attentionstore: Cost-effective atten- tion reuse across multi-turn conversations in large lan- guage model serving.arXiv preprint arXiv:2403.19708, 52:20–38","venue":"cs.CL","work_id":"d40a491e-677b-4dff-a6cb-636a8034acbc","year":2024},"citing_paper":{"arxiv_id":"2606.12556","last_updated":"2026-06-16T05:18:15Z","snapshot_observed_at":"2026-08-01T05:54:13.696296Z","submitted_at":"2026-06-10T18:06:30Z","title":"ITME: Inference Tiered Memory Expansion with Disaggregated CXL-Hybrid Memories","version":2},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-06-27T08:11:57.929452Z"},"links":{"cited_paper":"/paper/2403.19708","citing_paper":"/paper/2606.12556"},"observation_digest":"sha256:f2813914e61655758eb523d87afa687f66a2cf2264aad9abe6ea27ccffa22064","observation_id":"595e1dc6-daa4-413f-851e-8f9ee8550ae6","resolution":{"observed_at":"2026-07-03T13:18:13.432502Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.19708","last_updated":"2024-06-30T23:50:38Z","snapshot_observed_at":"2026-08-03T11:15:07.121206Z","submitted_at":"2024-03-23T10:42:49Z","title":"Cost-Efficient Large Language Model Serving for Multi-turn Conversations with CachedAttention","version":3},"cited_work":{"arxiv_id":"2403.19708","doi":"10.48550/arxiv.2403.19708","metadata_source":"pith","pith_arxiv_id":"2403.19708","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Attentionstore: Cost-effective atten- tion reuse across multi-turn conversations in large lan- guage model serving.arXiv preprint arXiv:2403.19708, 52:20–38","venue":"cs.CL","work_id":"d40a491e-677b-4dff-a6cb-636a8034acbc","year":2024},"citing_paper":{"arxiv_id":"2606.28565","last_updated":"2026-07-02T09:24:03Z","snapshot_observed_at":"2026-08-03T10:20:30.500902Z","submitted_at":"2026-06-26T19:43:38Z","title":"KernelSight-LM: A Kernel-Level LLM Inference Simulator","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-06-30T00:48:19.207465Z"},"links":{"cited_paper":"/paper/2403.19708","citing_paper":"/paper/2606.28565"},"observation_digest":"sha256:64f2d731343b9d7f44335eec197759e9284c22f303c9c32aefe4d05f19d01554","observation_id":"1ff9b9ca-560b-4466-99b2-d36e0b56ea61","resolution":{"observed_at":"2026-06-30T00:54:06.292563Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.19708","last_updated":"2024-06-30T23:50:38Z","snapshot_observed_at":"2026-08-03T11:15:07.121206Z","submitted_at":"2024-03-23T10:42:49Z","title":"Cost-Efficient Large Language Model Serving for Multi-turn Conversations with CachedAttention","version":3},"cited_work":{"arxiv_id":"2403.19708","doi":"10.48550/arxiv.2403.19708","metadata_source":"pith","pith_arxiv_id":"2403.19708","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Attentionstore: Cost-effective atten- tion reuse across multi-turn conversations in large lan- guage model serving.arXiv preprint arXiv:2403.19708, 52:20–38","venue":"cs.CL","work_id":"d40a491e-677b-4dff-a6cb-636a8034acbc","year":2024},"citing_paper":{"arxiv_id":"2606.28565","last_updated":"2026-07-02T09:24:03Z","snapshot_observed_at":"2026-08-03T10:20:30.500902Z","submitted_at":"2026-06-26T19:43:38Z","title":"KernelSight-LM: A Kernel-Level LLM Inference Simulator","version":2},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-07-03T23:09:38.092583Z"},"links":{"cited_paper":"/paper/2403.19708","citing_paper":"/paper/2606.28565"},"observation_digest":"sha256:a01740a8116678a128265460b61c13d232210ff03d60a24736d84f0a4ed9cfb0","observation_id":"fbf19b2e-d612-4291-95d5-fd9aae4c0e39","resolution":{"observed_at":"2026-07-03T23:19:02.272040Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.19708","last_updated":"2024-06-30T23:50:38Z","snapshot_observed_at":"2026-08-03T11:15:07.121206Z","submitted_at":"2024-03-23T10:42:49Z","title":"Cost-Efficient Large Language Model Serving for Multi-turn Conversations with CachedAttention","version":3},"cited_work":{"arxiv_id":"2403.19708","doi":"10.48550/arxiv.2403.19708","metadata_source":"pith","pith_arxiv_id":"2403.19708","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Attentionstore: Cost-effective atten- tion reuse across multi-turn conversations in large lan- guage model serving.arXiv preprint arXiv:2403.19708, 52:20–38","venue":"cs.CL","work_id":"d40a491e-677b-4dff-a6cb-636a8034acbc","year":2024},"citing_paper":{"arxiv_id":"2607.08032","last_updated":"2026-07-09T01:15:03Z","snapshot_observed_at":"2026-08-02T16:38:21.894642Z","submitted_at":"2026-07-09T01:15:03Z","title":"What to Keep, What to Forget: A Rate--Distortion View of Memory Compaction in LLMs and Agents","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-07-10T01:26:59.421158Z"},"links":{"cited_paper":"/paper/2403.19708","citing_paper":"/paper/2607.08032"},"observation_digest":"sha256:0a27ff0f2438f99f1cefcec88cb51f262cdd52159b8392a2cfdab8acba660e98","observation_id":"49cb944e-0b29-42de-8f4d-934d278c33a2","resolution":{"observed_at":"2026-07-10T01:36:44.007406Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.19708","last_updated":"2024-06-30T23:50:38Z","snapshot_observed_at":"2026-08-03T11:15:07.121206Z","submitted_at":"2024-03-23T10:42:49Z","title":"Cost-Efficient Large Language Model Serving for Multi-turn Conversations with CachedAttention","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.19708","snapshot_observed_at":"2026-08-01T15:58:25.475059Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.18141","last_updated":"2026-07-24T21:53:29Z","snapshot_observed_at":"2026-08-03T18:41:21.226842Z","submitted_at":"2026-07-20T16:35:47Z","title":"HyMCache: A KV Cache Framework for Multi-Turn LLM Serving with CXL-Hybrid Memory","version":2},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-01T15:58:25.475059Z"},"links":{"cited_paper":"/paper/2403.19708","citing_paper":"/paper/2607.18141"},"observation_digest":"sha256:b1b0796eddc76263edf6cfcf73476a4d6108d84239c1880156cdb7b8d47fcc05","observation_id":"9f592382-7ce0-4cf9-965a-1f805bb5f982","resolution":{"observed_at":"2026-08-01T15:58:25.475059Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2403.19708/citation-record","integrity":"/paper/2403.19708/integrity","json":"/paper/2403.19708/citation-record.json","paper":"/paper/2403.19708"},"outbound":[],"paper":{"arxiv_id":"2403.19708","last_updated":"2024-06-30T23:50:38Z","latest_version":3,"primary_category":"cs.CL","snapshot_observed_at":"2026-08-03T11:15:07.121206Z","submitted_at":"2024-03-23T10:42:49Z","title":"Cost-Efficient Large Language Model Serving for Multi-turn Conversations with CachedAttention"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"thesis":"As of 6 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 14 inbound Pith citation observations for arXiv:2403.19708."}