{"as_of":"2026-08-07T08:16:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:d914c5ba1540162e06707415dd0d1cbf905fbbc1848b6336acf361312f9cd043","coverage":[{"denominator":12,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":12,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-06T18:11:16.388804Z","state":"measured"},{"denominator":17,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":17,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-07T06:34:17.273281+00:00","state":"measured"},{"denominator":5,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":5,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-03T15:10:12.647037Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-06-28T23:52:49.103329Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2507.09019","last_updated":"2025-07-11T20:58:21Z","snapshot_observed_at":"2026-08-06T18:03:49.435059Z","submitted_at":"2025-07-11T20:58:21Z","title":"On Evaluating Performance of LLM Inference Serving Systems","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2507.09019","snapshot_observed_at":"2026-08-03T15:10:12.647037Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2512.18020","last_updated":"2025-12-19T19:24:56Z","snapshot_observed_at":"2026-08-05T16:53:40.020414Z","submitted_at":"2025-12-19T19:24:56Z","title":"Specification and Detection of LLM Code Smells","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-03T15:10:12.647037Z"},"links":{"cited_paper":"/paper/2507.09019","citing_paper":"/paper/2512.18020"},"observation_digest":"sha256:c92073733fdde0333921819e31081cdfcc9d25735b1a798a921d3332d82a37a2","observation_id":"d3452ba6-8358-48f7-8b0e-968225f2155d","resolution":{"observed_at":"2026-08-03T15:10:12.647037Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2507.09019","last_updated":"2025-07-11T20:58:21Z","snapshot_observed_at":"2026-08-06T18:03:49.435059Z","submitted_at":"2025-07-11T20:58:21Z","title":"On Evaluating Performance of LLM Inference Serving Systems","version":1},"cited_work":{"arxiv_id":"2507.09019","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2507.09019","snapshot_observed_at":"2026-06-28T23:52:49.103329Z","title":"On evaluating performance of llm inference serving systems","venue":null,"work_id":"16b1fb87-c364-4976-acde-225b20ac1e35","year":2025},"citing_paper":{"arxiv_id":"2602.01785","last_updated":"2026-04-28T16:05:53Z","snapshot_observed_at":"2026-07-06T22:44:04.951815Z","submitted_at":"2026-02-02T08:10:21Z","title":"CodeOCR: On the Effectiveness of Vision Language Models in Code Understanding","version":2},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-16T08:30:50.984873Z"},"links":{"cited_paper":"/paper/2507.09019","citing_paper":"/paper/2602.01785"},"observation_digest":"sha256:32221c8578c25bbdbc6554af4cd50f84ab544adbfe0f77905b8be0401cf1a775","observation_id":"e853f491-9ddf-45e8-957b-e4d986b6e523","resolution":{"observed_at":"2026-05-16T08:32:36.437742Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.09019","last_updated":"2025-07-11T20:58:21Z","snapshot_observed_at":"2026-08-06T18:03:49.435059Z","submitted_at":"2025-07-11T20:58:21Z","title":"On Evaluating Performance of LLM Inference Serving Systems","version":1},"cited_work":{"arxiv_id":"2507.09019","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2507.09019","snapshot_observed_at":"2026-06-28T23:52:49.103329Z","title":"On evaluating performance of llm inference serving systems","venue":null,"work_id":"16b1fb87-c364-4976-acde-225b20ac1e35","year":2025},"citing_paper":{"arxiv_id":"2604.09611","last_updated":"2026-03-12T10:10:37Z","snapshot_observed_at":"2026-08-06T15:21:40.163401Z","submitted_at":"2026-03-12T10:10:37Z","title":"Characterizing Performance-Energy Trade-offs of Large Language Models in Multi-Request Workflows","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-15T12:28:38.809950Z"},"links":{"cited_paper":"/paper/2507.09019","citing_paper":"/paper/2604.09611"},"observation_digest":"sha256:10fe79713e8ae8b7296f481ab14f0326232817a5854f0f08c9bee3c29ebae322","observation_id":"9ad232a9-fc3f-4fd8-b1f5-c2238c852d2a","resolution":{"observed_at":"2026-05-15T12:30:00.285701Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.09019","last_updated":"2025-07-11T20:58:21Z","snapshot_observed_at":"2026-08-06T18:03:49.435059Z","submitted_at":"2025-07-11T20:58:21Z","title":"On Evaluating Performance of LLM Inference Serving Systems","version":1},"cited_work":{"arxiv_id":"2507.09019","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2507.09019","snapshot_observed_at":"2026-06-28T23:52:49.103329Z","title":"On evaluating performance of llm inference serving systems","venue":null,"work_id":"16b1fb87-c364-4976-acde-225b20ac1e35","year":2025},"citing_paper":{"arxiv_id":"2604.15583","last_updated":"2026-04-24T15:54:27Z","snapshot_observed_at":"2026-07-06T23:03:07.228701Z","submitted_at":"2026-04-16T23:34:51Z","title":"SAGE: Selective Attention-Guided Extraction for Token-Efficient Document Indexing","version":2},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-10T08:32:02.222528Z"},"links":{"cited_paper":"/paper/2507.09019","citing_paper":"/paper/2604.15583"},"observation_digest":"sha256:3b5888783d6b611f277c0357a0acb49bec2a8e7eea2a76daa157fd66c1d5aa68","observation_id":"8598cbcc-8abd-43f3-973d-68d1c2d75d7f","resolution":{"observed_at":"2026-05-10T08:32:52.126778Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.09019","last_updated":"2025-07-11T20:58:21Z","snapshot_observed_at":"2026-08-06T18:03:49.435059Z","submitted_at":"2025-07-11T20:58:21Z","title":"On Evaluating Performance of LLM Inference Serving Systems","version":1},"cited_work":{"arxiv_id":"2507.09019","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2507.09019","snapshot_observed_at":"2026-06-28T23:52:49.103329Z","title":"On evaluating performance of llm inference serving systems","venue":null,"work_id":"16b1fb87-c364-4976-acde-225b20ac1e35","year":2025},"citing_paper":{"arxiv_id":"2605.29639","last_updated":"2026-05-28T09:07:06Z","snapshot_observed_at":"2026-08-03T07:14:27.185639Z","submitted_at":"2026-05-28T09:07:06Z","title":"RTP-LLM: High-Performance Alibaba LLM Inference Engine","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-06-28T23:52:40.763228Z"},"links":{"cited_paper":"/paper/2507.09019","citing_paper":"/paper/2605.29639"},"observation_digest":"sha256:2de2787d6fcf4a80993280bbc11e38ac046b0d782cefb8776ceb24afc499e5ce","observation_id":"9f3aa297-b3ea-47f7-8508-e3d155580047","resolution":{"observed_at":"2026-06-28T23:52:49.104862Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2507.09019/citation-record","integrity":"/paper/2507.09019/integrity","json":"/paper/2507.09019/citation-record.json","paper":"/paper/2507.09019"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2404.02015","last_updated":"2024-06-13T02:53:29Z","snapshot_observed_at":"2026-08-05T03:20:40.748106Z","submitted_at":"2024-04-02T14:56:43Z","title":"MuxServe: Flexible Spatial-Temporal Multiplexing for Multiple LLM Serving","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.02015","snapshot_observed_at":"2026-08-06T18:11:15.829163Z","title":"Muxserve: Flexible multiplexing for efficient multiple llm serving","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.09019","last_updated":"2025-07-11T20:58:21Z","snapshot_observed_at":"2026-08-06T18:03:49.435059Z","submitted_at":"2025-07-11T20:58:21Z","title":"On Evaluating Performance of LLM Inference Serving Systems","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-06T18:11:15.829163Z"},"links":{"cited_paper":"/paper/2404.02015","citing_paper":"/paper/2507.09019"},"observation_digest":"sha256:c7636af8f1d82d745c2594e4eef29b6d5ba5ef9f21add20b0809272707c4e545","observation_id":"28100343-384c-4f3d-b8fb-69ae1e05919f","resolution":{"observed_at":"2026-08-06T18:11:15.829163Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.07240","last_updated":"2024-07-19T21:04:14Z","snapshot_observed_at":"2026-07-06T16:31:03.559653Z","submitted_at":"2023-10-11T07:08:20Z","title":"CacheGen: KV Cache Compression and Streaming for Fast Large Language Model Serving","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.07240","snapshot_observed_at":"2026-08-06T18:11:15.906010Z","title":"Cachegen: Fast context loading for language model applications","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.09019","last_updated":"2025-07-11T20:58:21Z","snapshot_observed_at":"2026-08-06T18:03:49.435059Z","submitted_at":"2025-07-11T20:58:21Z","title":"On Evaluating Performance of LLM Inference Serving Systems","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-06T18:11:15.906010Z"},"links":{"cited_paper":"/paper/2310.07240","citing_paper":"/paper/2507.09019"},"observation_digest":"sha256:196ada9cf6bc71df4652c41d7746a2b4920106743bdb4e89ca203462f0bbdca5","observation_id":"aa5dc71d-861b-48c0-9a3a-05f55e5e1dc4","resolution":{"observed_at":"2026-08-06T18:11:15.906010Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.00079","last_updated":"2025-09-03T14:56:29Z","snapshot_observed_at":"2026-07-06T18:38:42.333605Z","submitted_at":"2024-06-24T02:05:32Z","title":"Mooncake: A KVCache-centric Disaggregated Architecture for LLM Serving","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.00079","snapshot_observed_at":"2026-08-06T18:11:15.969489Z","title":"Noam Shazeer","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.09019","last_updated":"2025-07-11T20:58:21Z","snapshot_observed_at":"2026-08-06T18:03:49.435059Z","submitted_at":"2025-07-11T20:58:21Z","title":"On Evaluating Performance of LLM Inference Serving Systems","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-06T18:11:15.969489Z"},"links":{"cited_paper":"/paper/2407.00079","citing_paper":"/paper/2507.09019"},"observation_digest":"sha256:fb186888ba58ca8f27b825dc0d0e730b2e0810e5dc02ea73b04f037cf7c942bb","observation_id":"9ebb2090-66ea-44ff-9cb0-72c5a1ba1f84","resolution":{"observed_at":"2026-08-06T18:11:15.969489Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:11:16.246971Z","title":"Guan Wang, Sijie Cheng, Xianyuan Zhan, Xiangang Li, Sen Song, and Yang Liu","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.09019","last_updated":"2025-07-11T20:58:21Z","snapshot_observed_at":"2026-08-06T18:03:49.435059Z","submitted_at":"2025-07-11T20:58:21Z","title":"On Evaluating Performance of LLM Inference Serving Systems","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-06T18:11:16.246971Z"},"links":{"citing_paper":"/paper/2507.09019"},"observation_digest":"sha256:f30a2670d8273d2aa109c83ae85f7b7beb65b7bb0ddd7c3ca5eb609651d5a872","observation_id":"dde8cc21-f904-417b-b360-f882410d750e","resolution":{"observed_at":"2026-08-06T18:11:16.246971Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.09526","last_updated":"2024-10-29T13:04:42Z","snapshot_observed_at":"2026-07-06T18:00:07.537987Z","submitted_at":"2024-04-15T07:45:04Z","title":"LoongServe: Efficiently Serving Long-Context Large Language Models with Elastic Sequence Parallelism","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.09526","snapshot_observed_at":"2026-08-06T18:11:16.254256Z","title":"Loongserve: Efficiently serving long-context large language models with elastic sequence parallelism","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.09019","last_updated":"2025-07-11T20:58:21Z","snapshot_observed_at":"2026-08-06T18:03:49.435059Z","submitted_at":"2025-07-11T20:58:21Z","title":"On Evaluating Performance of LLM Inference Serving Systems","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-06T18:11:16.254256Z"},"links":{"cited_paper":"/paper/2404.09526","citing_paper":"/paper/2507.09019"},"observation_digest":"sha256:205c15b76be0f5ff330d5eb5bd6ed905b1b2ab4a5f0884aabe5b72ce1c07c55a","observation_id":"9182bd48-f2ae-424d-a307-cd01d9864f44","resolution":{"observed_at":"2026-08-06T18:11:16.254256Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.12757","last_updated":"2025-05-25T14:08:01Z","snapshot_observed_at":"2026-08-03T08:33:17.421277Z","submitted_at":"2024-08-22T23:00:40Z","title":"NanoFlow: Towards Optimal Large Language Model Serving Throughput","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.12757","snapshot_observed_at":"2026-08-06T18:11:16.262684Z","title":"Nanoflow: Towards optimal large language model serving throughput","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.09019","last_updated":"2025-07-11T20:58:21Z","snapshot_observed_at":"2026-08-06T18:03:49.435059Z","submitted_at":"2025-07-11T20:58:21Z","title":"On Evaluating Performance of LLM Inference Serving Systems","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-06T18:11:16.262684Z"},"links":{"cited_paper":"/paper/2408.12757","citing_paper":"/paper/2507.09019"},"observation_digest":"sha256:8c40cea19d5ee067a58ec27d62ea4d73f2b5574f8247b36f4ca98085fdeab47d","observation_id":"11274d95-91cf-475e-b1a9-373b85f355a5","resolution":{"observed_at":"2026-08-06T18:11:16.262684Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.12757","last_updated":"2025-05-25T14:08:01Z","snapshot_observed_at":"2026-08-03T08:33:17.421277Z","submitted_at":"2024-08-22T23:00:40Z","title":"NanoFlow: Towards Optimal Large Language Model Serving Throughput","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.12757","snapshot_observed_at":"2026-08-06T18:11:16.322242Z","title":"URL https://doi.org/10.48550/arXiv.2408.12757","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.09019","last_updated":"2025-07-11T20:58:21Z","snapshot_observed_at":"2026-08-06T18:03:49.435059Z","submitted_at":"2025-07-11T20:58:21Z","title":"On Evaluating Performance of LLM Inference Serving Systems","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-06T18:11:16.322242Z"},"links":{"cited_paper":"/paper/2408.12757","citing_paper":"/paper/2507.09019"},"observation_digest":"sha256:8182292d6811e40188c146ff069ac5dec304270b05c5a391c33193e13c2a1f5c","observation_id":"55532324-5a12-45f1-b6fd-75421016dbd9","resolution":{"observed_at":"2026-08-06T18:11:16.322242Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T18:11:16.685457Z","title":"Parameter T uning (✓) In our judgment, the baselines don’t need parameter tuning","venue":null,"work_id":"dc1970ae-551f-4d54-8da7-7bec9d5b57bc","year":2023},"citing_paper":{"arxiv_id":"2507.09019","last_updated":"2025-07-11T20:58:21Z","snapshot_observed_at":"2026-08-06T18:03:49.435059Z","submitted_at":"2025-07-11T20:58:21Z","title":"On Evaluating Performance of LLM Inference Serving Systems","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-06T18:11:16.388804Z"},"links":{"citing_paper":"/paper/2507.09019"},"observation_digest":"sha256:57fec45c380b92b5b30a36e5100ffe4f994dcd21f2eb8f4359c8404939334aff","observation_id":"a9a0e147-24c3-4da7-9214-a9280e669f1b","resolution":{"observed_at":"2026-08-06T18:11:16.740082Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.03285","last_updated":"2024-06-05T06:06:43Z","snapshot_observed_at":"2026-07-06T16:43:42.741094Z","submitted_at":"2023-11-06T17:26:17Z","title":"S-LoRA: Serving Thousands of Concurrent LoRA Adapters","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.03285","snapshot_observed_at":"2026-08-06T18:11:16.146190Z","title":"S-lora: Serving thousands of concurrent lora adapters","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.09019","last_updated":"2025-07-11T20:58:21Z","snapshot_observed_at":"2026-08-06T18:03:49.435059Z","submitted_at":"2025-07-11T20:58:21Z","title":"On Evaluating Performance of LLM Inference Serving Systems","version":1},"reference_index":2017,"source":"pdf_text","source_observed_at":"2026-08-06T18:11:16.146190Z"},"links":{"cited_paper":"/paper/2311.03285","citing_paper":"/paper/2507.09019"},"observation_digest":"sha256:ebf2191846132857b3ca4c1cb406a073c6fa09f9fbda8f5b44239855a64be875","observation_id":"c5d486c1-2b71-4d34-973c-666392469670","resolution":{"observed_at":"2026-08-06T18:11:16.146190Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1701.06538","last_updated":"2017-01-23T18:10:00Z","snapshot_observed_at":"2026-07-06T05:27:13.416519Z","submitted_at":"2017-01-23T18:10:00Z","title":"Outrageously Large Neural Networks: The Sparsely-Gated Mixture-of-Experts Layer","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1701.06538","snapshot_observed_at":"2026-08-06T18:11:16.045289Z","title":"Outrageously large neural networks: The sparsely-gated mixture- of-experts layer","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.09019","last_updated":"2025-07-11T20:58:21Z","snapshot_observed_at":"2026-08-06T18:03:49.435059Z","submitted_at":"2025-07-11T20:58:21Z","title":"On Evaluating Performance of LLM Inference Serving Systems","version":1},"reference_index":2019,"source":"pdf_text","source_observed_at":"2026-08-06T18:11:16.045289Z"},"links":{"cited_paper":"/paper/1701.06538","citing_paper":"/paper/2507.09019"},"observation_digest":"sha256:6a3caa9ff79121438490010e5e3e1401f1fd1936515649b0724c8b80ead1233a","observation_id":"6861722a-35ce-4626-ab63-0f76e1044013","resolution":{"observed_at":"2026-08-06T18:11:16.045289Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2302.01318","last_updated":"2023-02-02T18:44:11Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-02-02T18:44:11Z","title":"Accelerating Large Language Model Decoding with Speculative Sampling","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2302.01318","snapshot_observed_at":"2026-08-06T18:11:15.769967Z","title":"Vidur: A Large-Scale Simulation Framework For LLM Inference","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.09019","last_updated":"2025-07-11T20:58:21Z","snapshot_observed_at":"2026-08-06T18:03:49.435059Z","submitted_at":"2025-07-11T20:58:21Z","title":"On Evaluating Performance of LLM Inference Serving Systems","version":1},"reference_index":2023,"source":"pdf_text","source_observed_at":"2026-08-06T18:11:15.769967Z"},"links":{"cited_paper":"/paper/2302.01318","citing_paper":"/paper/2507.09019"},"observation_digest":"sha256:24252dc9e077817ed8e251e4d7ca16da77cfd694c1a8496c8a252cb1ea207096","observation_id":"284a4b5b-dfb1-44f0-a2bb-a24dd331d451","resolution":{"observed_at":"2026-08-06T18:11:15.769967Z","resolver_source":null,"status":"malformed_identifier"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.15792","last_updated":"2024-08-28T13:35:54Z","snapshot_observed_at":"2026-07-06T19:07:07.234443Z","submitted_at":"2024-08-28T13:35:54Z","title":"Efficient LLM Scheduling by Learning to Rank","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.15792","snapshot_observed_at":"2026-08-06T18:11:15.858434Z","title":"Efficient llm scheduling by learning to rank","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.09019","last_updated":"2025-07-11T20:58:21Z","snapshot_observed_at":"2026-08-06T18:03:49.435059Z","submitted_at":"2025-07-11T20:58:21Z","title":"On Evaluating Performance of LLM Inference Serving Systems","version":1},"reference_index":2024,"source":"pdf_text","source_observed_at":"2026-08-06T18:11:15.858434Z"},"links":{"cited_paper":"/paper/2408.15792","citing_paper":"/paper/2507.09019"},"observation_digest":"sha256:7350ae4e81fd97e0e38bbfe2d8b62f57e8ca5d7b748d8e91d67c7e2678bf7fcb","observation_id":"9afccbaa-727d-41b4-8f2c-29152b6b8767","resolution":{"observed_at":"2026-08-06T18:11:15.858434Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2507.09019","last_updated":"2025-07-11T20:58:21Z","latest_version":1,"primary_category":"cs.LG","snapshot_observed_at":"2026-08-06T18:03:49.435059Z","submitted_at":"2025-07-11T20:58:21Z","title":"On Evaluating Performance of LLM Inference Serving Systems"},"reference_resolution":{"displayed":12,"state_counts":{"malformed_identifier":1,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":10,"verified_exact":0,"verified_fuzzy":1},"total_outbound_references":12},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"thesis":"As of 7 August 2026, this Paper Citation Record lists 12 of 12 outbound references and 5 inbound Pith citation observations for arXiv:2507.09019."}