{"as_of":"2026-08-07T17:21:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:ecb08562b42c49b1192e702b90fe8a47cd70c00cc137874f30e413e54734e37d","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":16,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":16,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-07T06:34:17.273281+00:00","state":"measured"},{"denominator":16,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":16,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T13:32:30.932279Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-05-21T15:30:17.948710Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2403.11421","last_updated":"2024-03-18T02:30:23Z","snapshot_observed_at":"2026-08-06T20:15:05.193939Z","submitted_at":"2024-03-18T02:30:23Z","title":"FastDecode: High-Throughput GPU-Efficient LLM Serving using Heterogeneous Pipelines","version":1},"cited_work":{"arxiv_id":"2403.11421","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2403.11421","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"and Zhai, J","venue":null,"work_id":"66245f70-061f-42d6-94d6-62e7f9950b0d","year":2024},"citing_paper":{"arxiv_id":"2404.14294","last_updated":"2024-07-19T04:47:36Z","snapshot_observed_at":"2026-07-06T18:03:47.096406Z","submitted_at":"2024-04-22T15:53:08Z","title":"A Survey on Efficient Inference for Large Language Models","version":3},"reference_index":252,"source":"pdf_text","source_observed_at":"2026-05-15T02:39:33.007894Z"},"links":{"cited_paper":"/paper/2403.11421","citing_paper":"/paper/2404.14294"},"observation_digest":"sha256:9a8e5971aa9929677d1ccc6ffa98519389579d094facf352bd566e2dcc1a107d","observation_id":"1aabc17c-9418-4901-9a3f-caef676b27bc","resolution":{"observed_at":"2026-05-15T02:39:33.397582Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.11421","last_updated":"2024-03-18T02:30:23Z","snapshot_observed_at":"2026-08-06T20:15:05.193939Z","submitted_at":"2024-03-18T02:30:23Z","title":"FastDecode: High-Throughput GPU-Efficient LLM Serving using Heterogeneous Pipelines","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.11421","snapshot_observed_at":"2026-08-07T13:32:30.932279Z","title":"Fastdecode: High-throughput gpu-efficient llm serving using heterogeneous pipelines, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.21487","last_updated":"2025-05-27T17:54:07Z","snapshot_observed_at":"2026-08-07T13:24:37.243429Z","submitted_at":"2025-05-27T17:54:07Z","title":"Hardware-Efficient Attention for Fast Decoding","version":1},"reference_index":28,"source":"arxiv_source","source_observed_at":"2026-08-07T13:32:30.932279Z"},"links":{"cited_paper":"/paper/2403.11421","citing_paper":"/paper/2505.21487"},"observation_digest":"sha256:56c46eb2c38eef18cbee67ac25013c5648f042cbf21c7050431fcff0e9891173","observation_id":"f576932a-1cf8-4630-82db-d0af173eecb0","resolution":{"observed_at":"2026-08-07T13:32:30.932279Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.11421","last_updated":"2024-03-18T02:30:23Z","snapshot_observed_at":"2026-08-06T20:15:05.193939Z","submitted_at":"2024-03-18T02:30:23Z","title":"FastDecode: High-Throughput GPU-Efficient LLM Serving using Heterogeneous Pipelines","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.11421","snapshot_observed_at":"2026-08-07T10:30:34.224262Z","title":"Fastdecode: High-throughput gpu-efficient llm serving using heterogeneous pipelines.arXiv preprint arXiv:2403.11421,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.05333","last_updated":"2025-06-20T01:25:25Z","snapshot_observed_at":"2026-08-07T10:18:59.977399Z","submitted_at":"2025-06-05T17:59:24Z","title":"Kinetics: Rethinking Test-Time Scaling Laws","version":3},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-07T10:30:34.224262Z"},"links":{"cited_paper":"/paper/2403.11421","citing_paper":"/paper/2506.05333"},"observation_digest":"sha256:966973dce92a2695c3bcd11d178dfc1128b7e24524926f93f7a85dd23403cbc5","observation_id":"25305e44-e40b-428b-8584-0e4fcba07bfe","resolution":{"observed_at":"2026-08-07T10:30:34.224262Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.11421","last_updated":"2024-03-18T02:30:23Z","snapshot_observed_at":"2026-08-06T20:15:05.193939Z","submitted_at":"2024-03-18T02:30:23Z","title":"FastDecode: High-Throughput GPU-Efficient LLM Serving using Heterogeneous Pipelines","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.11421","snapshot_observed_at":"2026-08-07T10:26:32.741822Z","title":"FastDecode: High-Throughput GPU-Efficient LLM Serving using Heteroge- neous Pipelines","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.05508","last_updated":"2025-06-05T18:47:49Z","snapshot_observed_at":"2026-08-07T13:07:21.433278Z","submitted_at":"2025-06-05T18:47:49Z","title":"Beyond the Buzz: A Pragmatic Take on Inference Disaggregation","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-07T10:26:32.741822Z"},"links":{"cited_paper":"/paper/2403.11421","citing_paper":"/paper/2506.05508"},"observation_digest":"sha256:c6ecf152981d8d39b5934223852b92e05042813034d9a8ac3a2a568c4537ce89","observation_id":"1db91454-2a3d-4e31-908e-ff723db9d091","resolution":{"observed_at":"2026-08-07T10:26:32.741822Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.11421","last_updated":"2024-03-18T02:30:23Z","snapshot_observed_at":"2026-08-06T20:15:05.193939Z","submitted_at":"2024-03-18T02:30:23Z","title":"FastDecode: High-Throughput GPU-Efficient LLM Serving using Heterogeneous Pipelines","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.11421","snapshot_observed_at":"2026-08-07T05:37:52.754758Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.07533","last_updated":"2025-06-09T08:16:24Z","snapshot_observed_at":"2026-08-07T05:29:08.151206Z","submitted_at":"2025-06-09T08:16:24Z","title":"MoQAE: Mixed-Precision Quantization for Long-Context LLM Inference via Mixture of Quantization-Aware Experts","version":1},"reference_index":11,"source":"arxiv_source","source_observed_at":"2026-08-07T05:37:52.754758Z"},"links":{"cited_paper":"/paper/2403.11421","citing_paper":"/paper/2506.07533"},"observation_digest":"sha256:2c63bb9f80975ba9970a149804154012caf781a8d5af993fb58d9bb0b11729c5","observation_id":"1b096e79-904e-4e7a-9c35-dab5e3ecdf9f","resolution":{"observed_at":"2026-08-07T05:37:52.754758Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.11421","last_updated":"2024-03-18T02:30:23Z","snapshot_observed_at":"2026-08-06T20:15:05.193939Z","submitted_at":"2024-03-18T02:30:23Z","title":"FastDecode: High-Throughput GPU-Efficient LLM Serving using Heterogeneous Pipelines","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.11421","snapshot_observed_at":"2026-08-07T12:38:37.472671Z","title":"Fastdecode: High-throughput gpu-efficient llm serving using heterogeneous pipelines","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.15704","last_updated":"2025-05-30T02:35:59Z","snapshot_observed_at":"2026-08-07T12:30:17.322980Z","submitted_at":"2025-05-30T02:35:59Z","title":"Learn from the Past: Fast Sparse Indexing for Large Language Model Decoding","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-07T12:38:37.472671Z"},"links":{"cited_paper":"/paper/2403.11421","citing_paper":"/paper/2506.15704"},"observation_digest":"sha256:18f3fb5d854eef32c1c99f941da4cadbded076f1dac614258fcc8e241b9fd4f4","observation_id":"a9c97e9d-4fe5-4fba-ba4d-0a63006dc951","resolution":{"observed_at":"2026-08-07T12:38:37.472671Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.11421","last_updated":"2024-03-18T02:30:23Z","snapshot_observed_at":"2026-08-06T20:15:05.193939Z","submitted_at":"2024-03-18T02:30:23Z","title":"FastDecode: High-Throughput GPU-Efficient LLM Serving using Heterogeneous Pipelines","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.11421","snapshot_observed_at":"2026-08-06T21:36:32.334492Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.23762","last_updated":"2025-06-30T12:09:29Z","snapshot_observed_at":"2026-08-06T21:29:49.550137Z","submitted_at":"2025-06-30T12:09:29Z","title":"Software Engineering for Large Language Models: Research Status, Challenges and the Road Ahead","version":1},"reference_index":123,"source":"pdf_text","source_observed_at":"2026-08-06T21:36:32.334492Z"},"links":{"cited_paper":"/paper/2403.11421","citing_paper":"/paper/2506.23762"},"observation_digest":"sha256:c2def34ae9d29a8f642d61ca9ba8db5d39a5a43fc5116b1e70135c7daa7dc57b","observation_id":"5737006c-aef0-4827-b38a-d548350d40ee","resolution":{"observed_at":"2026-08-06T21:36:32.334492Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.11421","last_updated":"2024-03-18T02:30:23Z","snapshot_observed_at":"2026-08-06T20:15:05.193939Z","submitted_at":"2024-03-18T02:30:23Z","title":"FastDecode: High-Throughput GPU-Efficient LLM Serving using Heterogeneous Pipelines","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.11421","snapshot_observed_at":"2026-08-06T17:20:26.333389Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.11273","last_updated":"2025-07-15T12:52:12Z","snapshot_observed_at":"2026-08-06T17:09:41.433446Z","submitted_at":"2025-07-15T12:52:12Z","title":"KV-Latent: Dimensional-level KV Cache Reduction with Frequency-aware Rotary Positional Embedding","version":1},"reference_index":15,"source":"arxiv_source","source_observed_at":"2026-08-06T17:20:26.333389Z"},"links":{"cited_paper":"/paper/2403.11421","citing_paper":"/paper/2507.11273"},"observation_digest":"sha256:2a75362c612d720bbb88be118cfe2fe509a4bf9e1301e68c0af67fb9e11e40f2","observation_id":"3104cf80-a1c0-4618-84aa-fcf5c0feb196","resolution":{"observed_at":"2026-08-06T17:20:26.333389Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.11421","last_updated":"2024-03-18T02:30:23Z","snapshot_observed_at":"2026-08-06T20:15:05.193939Z","submitted_at":"2024-03-18T02:30:23Z","title":"FastDecode: High-Throughput GPU-Efficient LLM Serving using Heterogeneous Pipelines","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.11421","snapshot_observed_at":"2026-08-06T14:06:37.907873Z","title":"Fastdecode: High-throughput gpu-efficient llm serving using heterogeneous pipelines","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.19823","last_updated":"2025-07-26T06:43:14Z","snapshot_observed_at":"2026-08-06T14:06:37.459069Z","submitted_at":"2025-07-26T06:43:14Z","title":"HCAttention: Extreme KV Cache Compression via Heterogeneous Attention Computing for LLMs","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-06T14:06:37.907873Z"},"links":{"cited_paper":"/paper/2403.11421","citing_paper":"/paper/2507.19823"},"observation_digest":"sha256:e0e10dcc9452fda08e201ca94c4f42d52dd97b21917201a38fd8c8bacec6d94a","observation_id":"950360d6-18a2-4927-9922-c48f3179ece1","resolution":{"observed_at":"2026-08-06T14:06:37.907873Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.11421","last_updated":"2024-03-18T02:30:23Z","snapshot_observed_at":"2026-08-06T20:15:05.193939Z","submitted_at":"2024-03-18T02:30:23Z","title":"FastDecode: High-Throughput GPU-Efficient LLM Serving using Heterogeneous Pipelines","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.11421","snapshot_observed_at":"2026-08-04T20:53:31.847951Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.08309","last_updated":"2025-09-10T06:06:51Z","snapshot_observed_at":"2026-08-06T05:24:56.655402Z","submitted_at":"2025-09-10T06:06:51Z","title":"Hetis: Serving LLMs in Heterogeneous GPU Clusters with Fine-grained and Dynamic Parallelism","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-04T20:53:31.847951Z"},"links":{"cited_paper":"/paper/2403.11421","citing_paper":"/paper/2509.08309"},"observation_digest":"sha256:cde397d52a9e1bbc676f88eed4c66750967f91a92a8667f3e6f23aadd4c5ee82","observation_id":"3e8225a2-deb0-4c9f-b07a-c421677e2b82","resolution":{"observed_at":"2026-08-04T20:53:31.847951Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.11421","last_updated":"2024-03-18T02:30:23Z","snapshot_observed_at":"2026-08-06T20:15:05.193939Z","submitted_at":"2024-03-18T02:30:23Z","title":"FastDecode: High-Throughput GPU-Efficient LLM Serving using Heterogeneous Pipelines","version":1},"cited_work":{"arxiv_id":"2403.11421","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2403.11421","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"and Zhai, J","venue":null,"work_id":"66245f70-061f-42d6-94d6-62e7f9950b0d","year":2024},"citing_paper":{"arxiv_id":"2601.20309","last_updated":"2026-05-18T19:51:16Z","snapshot_observed_at":"2026-07-06T22:43:17.026472Z","submitted_at":"2026-01-28T07:01:46Z","title":"SuperInfer: SLO-Aware Rotary Scheduling and Memory Management for LLM Inference on Superchips","version":2},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-05-21T15:26:01.283448Z"},"links":{"cited_paper":"/paper/2403.11421","citing_paper":"/paper/2601.20309"},"observation_digest":"sha256:00ad5750f7efca030c2894ca19eef45e47c08419a77e5145ae9b9dfd64cc7881","observation_id":"2f6b9d54-a29c-4d28-973a-c09b00bd5bb0","resolution":{"observed_at":"2026-05-21T15:30:17.950957Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.11421","last_updated":"2024-03-18T02:30:23Z","snapshot_observed_at":"2026-08-06T20:15:05.193939Z","submitted_at":"2024-03-18T02:30:23Z","title":"FastDecode: High-Throughput GPU-Efficient LLM Serving using Heterogeneous Pipelines","version":1},"cited_work":{"arxiv_id":"2403.11421","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2403.11421","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"and Zhai, J","venue":null,"work_id":"66245f70-061f-42d6-94d6-62e7f9950b0d","year":2024},"citing_paper":{"arxiv_id":"2601.22002","last_updated":"2026-08-01T18:15:51Z","snapshot_observed_at":"2026-08-06T23:10:55.941562Z","submitted_at":"2026-01-29T17:12:46Z","title":"Understanding Rate-Distortion Performance in Distributed Transformer Inference","version":3},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-05-16T10:01:53.390831Z"},"links":{"cited_paper":"/paper/2403.11421","citing_paper":"/paper/2601.22002"},"observation_digest":"sha256:bb6b167f5e2be94cee00b6bb122263db91b042d21c42179a08c0c3c690c22a28","observation_id":"90749329-89f8-4a3d-ba2b-73467a013d77","resolution":{"observed_at":"2026-05-16T10:02:42.597700Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.11421","last_updated":"2024-03-18T02:30:23Z","snapshot_observed_at":"2026-08-06T20:15:05.193939Z","submitted_at":"2024-03-18T02:30:23Z","title":"FastDecode: High-Throughput GPU-Efficient LLM Serving using Heterogeneous Pipelines","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.11421","snapshot_observed_at":"2026-08-03T06:51:13.983136Z","title":"Fastdecode: High-throughput GPU-efficient LLM serving using heterogeneous pipelines,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2601.22002","last_updated":"2026-08-01T18:15:51Z","snapshot_observed_at":"2026-08-06T23:10:55.941562Z","submitted_at":"2026-01-29T17:12:46Z","title":"Understanding Rate-Distortion Performance in Distributed Transformer Inference","version":4},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-03T06:51:13.983136Z"},"links":{"cited_paper":"/paper/2403.11421","citing_paper":"/paper/2601.22002"},"observation_digest":"sha256:10ba1b6bff72223cad0f7143121806b19ae1835155d1c6d099ae27e0ff409bba","observation_id":"8a7c8d74-18bd-448f-848f-31ac9a830ebc","resolution":{"observed_at":"2026-08-03T06:51:13.983136Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.11421","last_updated":"2024-03-18T02:30:23Z","snapshot_observed_at":"2026-08-06T20:15:05.193939Z","submitted_at":"2024-03-18T02:30:23Z","title":"FastDecode: High-Throughput GPU-Efficient LLM Serving using Heterogeneous Pipelines","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.11421","snapshot_observed_at":"2026-08-04T06:23:13.506645Z","title":"Fastdecode: High-throughput GPU-efficient LLM serving using heterogeneous pipelines,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2601.22002","last_updated":"2026-08-01T18:15:51Z","snapshot_observed_at":"2026-08-06T23:10:55.941562Z","submitted_at":"2026-01-29T17:12:46Z","title":"Understanding Rate-Distortion Performance in Distributed Transformer Inference","version":5},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-04T06:23:13.506645Z"},"links":{"cited_paper":"/paper/2403.11421","citing_paper":"/paper/2601.22002"},"observation_digest":"sha256:b3eddf668faaf2c3a00ac50d373766920c99e59d322b11f20c6f69c383d76287","observation_id":"0a05612f-965f-40c6-acb2-8d82e7d567af","resolution":{"observed_at":"2026-08-04T06:23:13.506645Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.11421","last_updated":"2024-03-18T02:30:23Z","snapshot_observed_at":"2026-08-06T20:15:05.193939Z","submitted_at":"2024-03-18T02:30:23Z","title":"FastDecode: High-Throughput GPU-Efficient LLM Serving using Heterogeneous Pipelines","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.11421","snapshot_observed_at":"2026-08-01T15:06:54.210093Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.26475","last_updated":"2026-07-29T05:01:42Z","snapshot_observed_at":"2026-08-07T15:06:08.198308Z","submitted_at":"2026-07-29T05:01:42Z","title":"DualDecoder: Accelerate Long Context LLM Inference by Predictive Prefetch","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-01T15:06:54.210093Z"},"links":{"cited_paper":"/paper/2403.11421","citing_paper":"/paper/2607.26475"},"observation_digest":"sha256:8530c08dfe098b6f19c99b33c6e87c44796dad03db4ee543034e548cc78af3e1","observation_id":"63dae347-e822-4f8a-80ad-d723fa99cf1c","resolution":{"observed_at":"2026-08-01T15:06:54.210093Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.11421","last_updated":"2024-03-18T02:30:23Z","snapshot_observed_at":"2026-08-06T20:15:05.193939Z","submitted_at":"2024-03-18T02:30:23Z","title":"FastDecode: High-Throughput GPU-Efficient LLM Serving using Heterogeneous Pipelines","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.11421","snapshot_observed_at":"2026-08-07T00:45:48.467163Z","title":"and Zhai, J","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.04405","last_updated":"2026-08-05T03:19:21Z","snapshot_observed_at":"2026-08-07T16:32:36.810418Z","submitted_at":"2026-08-05T03:19:21Z","title":"Training-Free Hashing-Based Attention via Binary Principal Components","version":1},"reference_index":53,"source":"arxiv_source","source_observed_at":"2026-08-07T00:45:48.467163Z"},"links":{"cited_paper":"/paper/2403.11421","citing_paper":"/paper/2608.04405"},"observation_digest":"sha256:83a3675a64d7c2fb74749bd1e389371fe6fc4ef6d8d0bc00753ec6e1b5d22c98","observation_id":"e3cec5c8-1849-4e2e-89b7-01434f9dcc5b","resolution":{"observed_at":"2026-08-07T00:45:48.467163Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2403.11421/citation-record","integrity":"/paper/2403.11421/integrity","json":"/paper/2403.11421/citation-record.json","paper":"/paper/2403.11421"},"outbound":[],"paper":{"arxiv_id":"2403.11421","last_updated":"2024-03-18T02:30:23Z","latest_version":1,"primary_category":"cs.DC","snapshot_observed_at":"2026-08-06T20:15:05.193939Z","submitted_at":"2024-03-18T02:30:23Z","title":"FastDecode: High-Throughput GPU-Efficient LLM Serving using Heterogeneous Pipelines"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"thesis":"As of 7 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 16 inbound Pith citation observations for arXiv:2403.11421."}