{"as_of":"2026-08-15T13:41:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:d1dcc92e617b85bba396ba88a9ca61c4774d4bdaa211afbd55a7b2cfd2d8868e","coverage":[{"denominator":71,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":71,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-14T05:50:05.925477Z","state":"measured"},{"denominator":71,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":71,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-15T06:32:42.880941+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2608.13499/citation-record","integrity":"/paper/2608.13499/integrity","json":"/paper/2608.13499/citation-record.json","paper":"/paper/2608.13499"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T05:50:07.553134Z","title":"Taming Throughput-Latency Tradeoff in LLM Inference with Sarathi-Serve","venue":null,"work_id":"0b6b8c95-40db-45ad-bb89-addf08941e6c","year":2024},"citing_paper":{"arxiv_id":"2608.13499","last_updated":"2026-08-13T17:28:54Z","snapshot_observed_at":"2026-08-15T13:16:28.671545Z","submitted_at":"2026-08-13T17:28:54Z","title":"OpScale: Operator-level Provisioning and Autoscaling for LLM Serving","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-14T05:50:05.568677Z"},"links":{"citing_paper":"/paper/2608.13499"},"observation_digest":"sha256:2010a13b8a664b373c19e8ad48594e43779c4f752e7d17a513b976aedf9438e3","observation_id":"d5b0165d-3fde-44d8-a308-1c4a95bd7cbf","resolution":{"observed_at":"2026-08-14T05:50:07.558975Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T05:50:05.574731Z","title":"Medha: Efficiently serving multi-million context length LLM inference requests without approximations.arXiv preprint arXiv:2409.17264, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.13499","last_updated":"2026-08-13T17:28:54Z","snapshot_observed_at":"2026-08-15T13:16:28.671545Z","submitted_at":"2026-08-13T17:28:54Z","title":"OpScale: Operator-level Provisioning and Autoscaling for LLM Serving","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-14T05:50:05.574731Z"},"links":{"citing_paper":"/paper/2608.13499"},"observation_digest":"sha256:9d8563384c9fd351ce464d84f12377f2d193121eb30c3dc23492e7d85566d345","observation_id":"5f6fc802-b4cf-4773-a999-bf8eac2017e3","resolution":{"observed_at":"2026-08-14T05:50:05.574731Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T05:50:07.536946Z","title":"Look Ma, No Bubbles! Designing a Low-Latency Megakernel for Llama-1B","venue":null,"work_id":"8ecfa8f5-d620-42a8-8bee-8a8f7cbf049a","year":2025},"citing_paper":{"arxiv_id":"2608.13499","last_updated":"2026-08-13T17:28:54Z","snapshot_observed_at":"2026-08-15T13:16:28.671545Z","submitted_at":"2026-08-13T17:28:54Z","title":"OpScale: Operator-level Provisioning and Autoscaling for LLM Serving","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-14T05:50:05.580018Z"},"links":{"citing_paper":"/paper/2608.13499"},"observation_digest":"sha256:aafe2d879bf1fb2811017cfefa57276b4cfe7bc3f361c0f714dfcf8b1a33b1a7","observation_id":"cd6e8b85-6033-4dbc-982f-e3bb0f6b1970","resolution":{"observed_at":"2026-08-14T05:50:07.542109Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T05:50:07.519710Z","title":"Internet and the Erlang formula.ACM SIGCOMM Computer Communication Review, 42(1):23–30, 2012","venue":null,"work_id":"7620df37-aea8-4949-ad00-f0e368be9f4e","year":2012},"citing_paper":{"arxiv_id":"2608.13499","last_updated":"2026-08-13T17:28:54Z","snapshot_observed_at":"2026-08-15T13:16:28.671545Z","submitted_at":"2026-08-13T17:28:54Z","title":"OpScale: Operator-level Provisioning and Autoscaling for LLM Serving","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-14T05:50:05.585184Z"},"links":{"citing_paper":"/paper/2608.13499"},"observation_digest":"sha256:a63386bfd1e18584a5f4950dd9cab34814a73ee125a3314bd3ca58f82dab4f62","observation_id":"16cda042-bf71-4f5f-b21b-40049ac7ae63","resolution":{"observed_at":"2026-08-14T05:50:07.525517Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T05:50:07.503341Z","title":"Stability, queue length, and delay of deterministic and stochastic queueing networks","venue":null,"work_id":"0855264d-88a2-44c8-9f90-07edf1e0c592","year":1994},"citing_paper":{"arxiv_id":"2608.13499","last_updated":"2026-08-13T17:28:54Z","snapshot_observed_at":"2026-08-15T13:16:28.671545Z","submitted_at":"2026-08-13T17:28:54Z","title":"OpScale: Operator-level Provisioning and Autoscaling for LLM Serving","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-14T05:50:05.590364Z"},"links":{"citing_paper":"/paper/2608.13499"},"observation_digest":"sha256:80c257fefbbc87236698f2b8c7531301c11ceaef9429e72d9a8165cd60fec044","observation_id":"ad8f9766-7a47-4bef-b45e-a748f1e1e46b","resolution":{"observed_at":"2026-08-14T05:50:07.509278Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T05:50:07.486947Z","title":"TVM: An automated end-to-end optimizing compiler for deep learning","venue":null,"work_id":"c1630a88-7161-415e-9b43-24d52bf9bd05","year":2018},"citing_paper":{"arxiv_id":"2608.13499","last_updated":"2026-08-13T17:28:54Z","snapshot_observed_at":"2026-08-15T13:16:28.671545Z","submitted_at":"2026-08-13T17:28:54Z","title":"OpScale: Operator-level Provisioning and Autoscaling for LLM Serving","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-14T05:50:05.595384Z"},"links":{"citing_paper":"/paper/2608.13499"},"observation_digest":"sha256:5a1efdb141ea821f133a073e7c316d71157e7af475a896934e26b1abea1e2243","observation_id":"68dbfb04-249d-4ec9-b604-d4cd9f6d3090","resolution":{"observed_at":"2026-08-14T05:50:07.493059Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T05:50:07.470167Z","title":"Towards high-goodput LLM serving with prefill-decode multiplexing","venue":null,"work_id":"1c4e1f1f-9c25-44a2-b37b-62b0f2304634","year":2026},"citing_paper":{"arxiv_id":"2608.13499","last_updated":"2026-08-13T17:28:54Z","snapshot_observed_at":"2026-08-15T13:16:28.671545Z","submitted_at":"2026-08-13T17:28:54Z","title":"OpScale: Operator-level Provisioning and Autoscaling for LLM Serving","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-14T05:50:05.601314Z"},"links":{"citing_paper":"/paper/2608.13499"},"observation_digest":"sha256:0f24f8151aac004892e21f3f4c0d5f1084fd784606495e9771ee7daf97d3857e","observation_id":"0fefdded-8a8b-4e27-905e-e3500f08e17f","resolution":{"observed_at":"2026-08-14T05:50:07.475563Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T05:50:05.607051Z","title":"Mirage Persistent Kernel: A Compiler and Runtime for Mega-Kernelizing Tensor Programs.arXiv preprint arXiv:2512.22219, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.13499","last_updated":"2026-08-13T17:28:54Z","snapshot_observed_at":"2026-08-15T13:16:28.671545Z","submitted_at":"2026-08-13T17:28:54Z","title":"OpScale: Operator-level Provisioning and Autoscaling for LLM Serving","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-14T05:50:05.607051Z"},"links":{"citing_paper":"/paper/2608.13499"},"observation_digest":"sha256:509e0550854167d46273200c13b834e1b59ed9fbbf231969c388bf86be107329","observation_id":"2baf0e86-eecd-4c90-afb2-57ff39e6db10","resolution":{"observed_at":"2026-08-14T05:50:05.607051Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T05:50:07.454759Z","title":"Serving heterogeneous machine learning models on multi-GPU servers with spatio-temporal sharing","venue":null,"work_id":"fa9a7bb9-d059-486e-8cca-3efd4faaf08d","year":2022},"citing_paper":{"arxiv_id":"2608.13499","last_updated":"2026-08-13T17:28:54Z","snapshot_observed_at":"2026-08-15T13:16:28.671545Z","submitted_at":"2026-08-13T17:28:54Z","title":"OpScale: Operator-level Provisioning and Autoscaling for LLM Serving","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-14T05:50:05.612248Z"},"links":{"citing_paper":"/paper/2608.13499"},"observation_digest":"sha256:7f8c57a0951608300e6e727064b972e732f3e99d178fd084abb8c53664122036","observation_id":"1cd6590e-3aac-4d18-a443-5bb3e3cab5f5","resolution":{"observed_at":"2026-08-14T05:50:07.460293Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T05:50:07.438457Z","title":"PaLM: Scaling language modeling with Pathways.Journal of Machine Learning Research, 24(240):1–113, 2023","venue":null,"work_id":"62a2d67e-4721-413b-b14d-d9ad1807d33f","year":2023},"citing_paper":{"arxiv_id":"2608.13499","last_updated":"2026-08-13T17:28:54Z","snapshot_observed_at":"2026-08-15T13:16:28.671545Z","submitted_at":"2026-08-13T17:28:54Z","title":"OpScale: Operator-level Provisioning and Autoscaling for LLM Serving","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-14T05:50:05.617457Z"},"links":{"citing_paper":"/paper/2608.13499"},"observation_digest":"sha256:9334cab83843d3083e20b6c559c60e7f81854db17b50bae9dfb278ca67c6b530","observation_id":"0d62dfb8-90d2-48b0-aaf8-8cff978ee44e","resolution":{"observed_at":"2026-08-14T05:50:07.443604Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T05:50:07.423396Z","title":"LithOS: An operating system for efficient machine learning on GPUs","venue":null,"work_id":"2ac24323-ee98-4a8e-9dda-82b257a5d6e6","year":2025},"citing_paper":{"arxiv_id":"2608.13499","last_updated":"2026-08-13T17:28:54Z","snapshot_observed_at":"2026-08-15T13:16:28.671545Z","submitted_at":"2026-08-13T17:28:54Z","title":"OpScale: Operator-level Provisioning and Autoscaling for LLM Serving","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-14T05:50:05.622750Z"},"links":{"citing_paper":"/paper/2608.13499"},"observation_digest":"sha256:4e87ea6af3d5ffc05387c8dc7c8d6282dbbf834f4c65020314a9fbc0641a4c1e","observation_id":"6c5e8fba-a593-4d5d-9b4c-e2f2e464a0a4","resolution":{"observed_at":"2026-08-14T05:50:07.428119Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T05:50:07.407791Z","title":"FlashAttention: Fast and memory-efficient exact attention with IO-awareness","venue":null,"work_id":"ca63a9b8-2218-4b58-acca-3110eddbc6a7","year":2022},"citing_paper":{"arxiv_id":"2608.13499","last_updated":"2026-08-13T17:28:54Z","snapshot_observed_at":"2026-08-15T13:16:28.671545Z","submitted_at":"2026-08-13T17:28:54Z","title":"OpScale: Operator-level Provisioning and Autoscaling for LLM Serving","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-14T05:50:05.628242Z"},"links":{"citing_paper":"/paper/2608.13499"},"observation_digest":"sha256:77345f3fab493d49a079fc7311f6b192caf8fc55752323ac97845fc457879576","observation_id":"a9f31d62-9979-44a0-91d1-1825d6ad419a","resolution":{"observed_at":"2026-08-14T05:50:07.413635Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T05:50:07.391584Z","title":"GSLICE: controlled spatial sharing of GPUs for a scalable inference platform","venue":null,"work_id":"b69e9747-97b2-434d-86aa-6f12e463f9da","year":2020},"citing_paper":{"arxiv_id":"2608.13499","last_updated":"2026-08-13T17:28:54Z","snapshot_observed_at":"2026-08-15T13:16:28.671545Z","submitted_at":"2026-08-13T17:28:54Z","title":"OpScale: Operator-level Provisioning and Autoscaling for LLM Serving","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-14T05:50:05.633194Z"},"links":{"citing_paper":"/paper/2608.13499"},"observation_digest":"sha256:74d0e9b5212d58aee8495a1d7718902fd077435eedb7131c288f239c4872b66d","observation_id":"95efc589-7ff8-4255-9460-c32d9e07069c","resolution":{"observed_at":"2026-08-14T05:50:07.396879Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T05:50:05.638466Z","title":"HydraInfer: Hybrid disaggregated scheduling for multimodal large language model serving.arXiv preprint arXiv:2505.12658, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.13499","last_updated":"2026-08-13T17:28:54Z","snapshot_observed_at":"2026-08-15T13:16:28.671545Z","submitted_at":"2026-08-13T17:28:54Z","title":"OpScale: Operator-level Provisioning and Autoscaling for LLM Serving","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-14T05:50:05.638466Z"},"links":{"citing_paper":"/paper/2608.13499"},"observation_digest":"sha256:c83c694ef3c34515d7f2de323138fa5fbed6cf4b08c11bb6302510c830b0e4e9","observation_id":"0a27caf4-6d0b-403a-b2e3-b570a1d146db","resolution":{"observed_at":"2026-08-14T05:50:05.638466Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.02015","last_updated":"2024-06-13T02:53:29Z","snapshot_observed_at":"2026-08-13T00:39:13.375770Z","submitted_at":"2024-04-02T14:56:43Z","title":"MuxServe: Flexible Spatial-Temporal Multiplexing for Multiple LLM Serving","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.02015","snapshot_observed_at":"2026-08-14T05:50:05.643366Z","title":"MuxServe: Flexible spatial-temporal multiplexing for multiple LLM serving.arXiv preprint arXiv:2404.02015, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.13499","last_updated":"2026-08-13T17:28:54Z","snapshot_observed_at":"2026-08-15T13:16:28.671545Z","submitted_at":"2026-08-13T17:28:54Z","title":"OpScale: Operator-level Provisioning and Autoscaling for LLM Serving","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-14T05:50:05.643366Z"},"links":{"cited_paper":"/paper/2404.02015","citing_paper":"/paper/2608.13499"},"observation_digest":"sha256:9b519670503e141f5d8c7aac8359a3354618509434a343d9c854cc83bae758cc","observation_id":"3c21e7f3-3e67-45ea-b50e-a33072e1a513","resolution":{"observed_at":"2026-08-14T05:50:05.643366Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T05:50:07.375985Z","title":"ServerlessLLM:low-latency serverless inference for large language models","venue":null,"work_id":"991a1a65-50fd-4b0e-abd6-03058005f62a","year":2024},"citing_paper":{"arxiv_id":"2608.13499","last_updated":"2026-08-13T17:28:54Z","snapshot_observed_at":"2026-08-15T13:16:28.671545Z","submitted_at":"2026-08-13T17:28:54Z","title":"OpScale: Operator-level Provisioning and Autoscaling for LLM Serving","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-14T05:50:05.648448Z"},"links":{"citing_paper":"/paper/2608.13499"},"observation_digest":"sha256:dc6cc5d13d5ca675cbc18bf7c2c9e1a4ab9ae8182d63cdd25c5ca534335c5f35","observation_id":"6642cb48-6559-4fd3-8ae6-560efea4baf9","resolution":{"observed_at":"2026-08-14T05:50:07.381562Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T05:50:07.361062Z","title":"ATOM: Model-driven autoscaling for microservices","venue":null,"work_id":"80a5f0bc-fa33-48e6-b8cb-56be9d5d0554","year":2019},"citing_paper":{"arxiv_id":"2608.13499","last_updated":"2026-08-13T17:28:54Z","snapshot_observed_at":"2026-08-15T13:16:28.671545Z","submitted_at":"2026-08-13T17:28:54Z","title":"OpScale: Operator-level Provisioning and Autoscaling for LLM Serving","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-14T05:50:05.653791Z"},"links":{"citing_paper":"/paper/2608.13499"},"observation_digest":"sha256:1fd38a1e63661f7a8ca2390fa8a6873e47e96b52cfe79450856f3c6f59bd01d6","observation_id":"4294a411-ecdb-46f5-a779-cc993332638b","resolution":{"observed_at":"2026-08-14T05:50:07.365712Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T05:50:07.344935Z","title":"Nano-vLLM.https: //github.com/GeeeekExplorer/nano-vllm, 2025","venue":null,"work_id":"a037ef8f-da77-4f86-8ba9-c41ebe091f3c","year":2025},"citing_paper":{"arxiv_id":"2608.13499","last_updated":"2026-08-13T17:28:54Z","snapshot_observed_at":"2026-08-15T13:16:28.671545Z","submitted_at":"2026-08-13T17:28:54Z","title":"OpScale: Operator-level Provisioning and Autoscaling for LLM Serving","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-14T05:50:05.658523Z"},"links":{"citing_paper":"/paper/2608.13499"},"observation_digest":"sha256:87c37c0b850df09cf6157d614679d7a0c8dd3a4275c0f988a0849227a40d35b7","observation_id":"938876dd-505b-4dd3-99e0-3fe6c9225c4c","resolution":{"observed_at":"2026-08-14T05:50:07.349857Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T05:50:07.326878Z","title":"NVIDIA Dynamo","venue":null,"work_id":"4d17a760-7f7e-4b9f-83d3-3a943320ab1d","year":2025},"citing_paper":{"arxiv_id":"2608.13499","last_updated":"2026-08-13T17:28:54Z","snapshot_observed_at":"2026-08-15T13:16:28.671545Z","submitted_at":"2026-08-13T17:28:54Z","title":"OpScale: Operator-level Provisioning and Autoscaling for LLM Serving","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-14T05:50:05.664438Z"},"links":{"citing_paper":"/paper/2608.13499"},"observation_digest":"sha256:e0e6e408a16c3d7241912e1aded0fd61f42c1a2d89131c7d4f4842f925264981","observation_id":"f60a81f9-0807-462d-9bb5-cb62b20f2324","resolution":{"observed_at":"2026-08-14T05:50:07.334485Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T05:50:07.309415Z","title":"vLLM Production Stack.https: //github.com/vllm-project/production-stack, 2025","venue":null,"work_id":"52e41d87-5e25-4d4d-8ae8-7b0c4ae1ab3d","year":2025},"citing_paper":{"arxiv_id":"2608.13499","last_updated":"2026-08-13T17:28:54Z","snapshot_observed_at":"2026-08-15T13:16:28.671545Z","submitted_at":"2026-08-13T17:28:54Z","title":"OpScale: Operator-level Provisioning and Autoscaling for LLM Serving","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-14T05:50:05.670158Z"},"links":{"citing_paper":"/paper/2608.13499"},"observation_digest":"sha256:9783e94e31fce5537b84e0a04d79a3678df6ab41411a51ecd3887b03872a5abc","observation_id":"43697b43-a2b8-49ae-92f0-db0b84ad1d31","resolution":{"observed_at":"2026-08-14T05:50:07.315794Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.19867","last_updated":"2025-04-28T15:00:03Z","snapshot_observed_at":"2026-08-13T22:19:53.894883Z","submitted_at":"2025-04-28T15:00:03Z","title":"semi-PD: Towards Efficient LLM Serving via Phase-Wise Disaggregated Computation and Unified Storage","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.19867","snapshot_observed_at":"2026-08-14T05:50:05.675450Z","title":"Semi-PD: Towards efficient LLM serving via phase-wise disaggregated computation and unified storage.arXiv preprint arXiv:2504.19867, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.13499","last_updated":"2026-08-13T17:28:54Z","snapshot_observed_at":"2026-08-15T13:16:28.671545Z","submitted_at":"2026-08-13T17:28:54Z","title":"OpScale: Operator-level Provisioning and Autoscaling for LLM Serving","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-14T05:50:05.675450Z"},"links":{"cited_paper":"/paper/2504.19867","citing_paper":"/paper/2608.13499"},"observation_digest":"sha256:e10ff470db737a4cb137a193107cb3c5f01eaeff45347767020980bc737e3e06","observation_id":"3c53029e-3559-4473-98af-3781038d5753","resolution":{"observed_at":"2026-08-14T05:50:05.675450Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T05:50:07.293163Z","title":"DEEPSERVE: Serverless large language model serving at scale","venue":null,"work_id":"e86d64a2-337b-42a2-bb00-2828a1b47c54","year":2025},"citing_paper":{"arxiv_id":"2608.13499","last_updated":"2026-08-13T17:28:54Z","snapshot_observed_at":"2026-08-15T13:16:28.671545Z","submitted_at":"2026-08-13T17:28:54Z","title":"OpScale: Operator-level Provisioning and Autoscaling for LLM Serving","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-14T05:50:05.680591Z"},"links":{"citing_paper":"/paper/2608.13499"},"observation_digest":"sha256:7b1c643517fcde4bcadd844bb760570130df44a33d460e45ccb4884c5df87e16","observation_id":"28ec9bbf-799b-4bea-917e-ed09ae1cef7d","resolution":{"observed_at":"2026-08-14T05:50:07.298852Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.13497","last_updated":"2025-06-16T13:54:41Z","snapshot_observed_at":"2026-08-13T12:12:46.051979Z","submitted_at":"2025-06-16T13:54:41Z","title":"DDiT: Dynamic Resource Allocation for Diffusion Transformer Model Serving","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2506.13497","snapshot_observed_at":"2026-08-14T05:50:05.685350Z","title":"DDiT: Dynamic resource allocation for diffusion transformer model serving.arXiv preprint arXiv:2506.13497, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.13499","last_updated":"2026-08-13T17:28:54Z","snapshot_observed_at":"2026-08-15T13:16:28.671545Z","submitted_at":"2026-08-13T17:28:54Z","title":"OpScale: Operator-level Provisioning and Autoscaling for LLM Serving","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-14T05:50:05.685350Z"},"links":{"cited_paper":"/paper/2506.13497","citing_paper":"/paper/2608.13499"},"observation_digest":"sha256:60b564bd2642fd138e1ecfee9a1b7b6d9bdd08a41cdacdbe12afef39df2debeb","observation_id":"b65cc737-39d8-4e89-85c3-5b3c43cb3795","resolution":{"observed_at":"2026-08-14T05:50:05.685350Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T05:50:07.277843Z","title":"In 2026 IEEE International Symposium on High Performance Computer Architecture (HPCA 2026), pages 1–14","venue":null,"work_id":"bf5f0ab4-874f-41ba-9106-be3e76e106e0","year":2026},"citing_paper":{"arxiv_id":"2608.13499","last_updated":"2026-08-13T17:28:54Z","snapshot_observed_at":"2026-08-15T13:16:28.671545Z","submitted_at":"2026-08-13T17:28:54Z","title":"OpScale: Operator-level Provisioning and Autoscaling for LLM Serving","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-14T05:50:05.691663Z"},"links":{"citing_paper":"/paper/2608.13499"},"observation_digest":"sha256:c939e67c7922e3f27a89d9aed603a441ba6be0f0f3deefc24c21d31655286ac3","observation_id":"499f5144-81ba-401c-b2fa-2ef8e3a065e2","resolution":{"observed_at":"2026-08-14T05:50:07.282903Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T05:50:07.258551Z","title":"Llama-3-8b.https://huggingface","venue":null,"work_id":"f071866f-8175-46d0-876b-61665cbf9965","year":2025},"citing_paper":{"arxiv_id":"2608.13499","last_updated":"2026-08-13T17:28:54Z","snapshot_observed_at":"2026-08-15T13:16:28.671545Z","submitted_at":"2026-08-13T17:28:54Z","title":"OpScale: Operator-level Provisioning and Autoscaling for LLM Serving","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-14T05:50:05.696322Z"},"links":{"citing_paper":"/paper/2608.13499"},"observation_digest":"sha256:e6d3ed9c3c0bd739c32f2037ff8cfded7933d3bbdef941a13e30c8486d60df7a","observation_id":"29fa7ea1-7334-464c-a2f2-b9314d28960a","resolution":{"observed_at":"2026-08-14T05:50:07.265449Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T05:50:07.240645Z","title":"Mixtral-8x7B-v0.1.https:// huggingface.co/mistralai/Mixtral-8x7B-v0.1, 2025","venue":null,"work_id":"b3e64a37-c7c5-4164-bdbe-0276322fc735","year":2025},"citing_paper":{"arxiv_id":"2608.13499","last_updated":"2026-08-13T17:28:54Z","snapshot_observed_at":"2026-08-15T13:16:28.671545Z","submitted_at":"2026-08-13T17:28:54Z","title":"OpScale: Operator-level Provisioning and Autoscaling for LLM Serving","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-14T05:50:05.701113Z"},"links":{"citing_paper":"/paper/2608.13499"},"observation_digest":"sha256:812f2691ae4d7c2b2d734d06a70c1ca351a1e8a02bdf5a7dd702eb7d18ed4f51","observation_id":"24b94bb3-2292-4c52-a28c-9c24e6a8b3bf","resolution":{"observed_at":"2026-08-14T05:50:07.246809Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T05:50:07.225417Z","title":"Qwen2-57B-A14B","venue":null,"work_id":"00b1c24f-fb72-4f21-b73c-0d83fc6786ac","year":2025},"citing_paper":{"arxiv_id":"2608.13499","last_updated":"2026-08-13T17:28:54Z","snapshot_observed_at":"2026-08-15T13:16:28.671545Z","submitted_at":"2026-08-13T17:28:54Z","title":"OpScale: Operator-level Provisioning and Autoscaling for LLM Serving","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-14T05:50:05.706060Z"},"links":{"citing_paper":"/paper/2608.13499"},"observation_digest":"sha256:17e02a7ef445e67f24d60aa8e58bf0c4fba270bbaae8a5a44fb76c1f356f6caa","observation_id":"e672d2cb-277e-4f5b-b8ef-9be6f7661577","resolution":{"observed_at":"2026-08-14T05:50:07.230407Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T05:50:07.209759Z","title":"QWen2-7B-Instruct.https: //huggingface.co/Qwen/Qwen2-7B-Instruct, 2025","venue":null,"work_id":"23ea7313-68af-4c44-b15a-707f777c46e3","year":2025},"citing_paper":{"arxiv_id":"2608.13499","last_updated":"2026-08-13T17:28:54Z","snapshot_observed_at":"2026-08-15T13:16:28.671545Z","submitted_at":"2026-08-13T17:28:54Z","title":"OpScale: Operator-level Provisioning and Autoscaling for LLM Serving","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-14T05:50:05.710659Z"},"links":{"citing_paper":"/paper/2608.13499"},"observation_digest":"sha256:c0b9d82a143f1cd413b6d5196b060f4dbdab21b291f0b972b7d3404de3128f9f","observation_id":"1b2ccc7f-1231-4695-855f-f2444195f0b6","resolution":{"observed_at":"2026-08-14T05:50:07.214383Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T05:50:07.195359Z","title":"Qwen2.5-VL-32B","venue":null,"work_id":"fe46eca1-f243-4f06-8d6f-01db90b4e220","year":2025},"citing_paper":{"arxiv_id":"2608.13499","last_updated":"2026-08-13T17:28:54Z","snapshot_observed_at":"2026-08-15T13:16:28.671545Z","submitted_at":"2026-08-13T17:28:54Z","title":"OpScale: Operator-level Provisioning and Autoscaling for LLM Serving","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-14T05:50:05.715410Z"},"links":{"citing_paper":"/paper/2608.13499"},"observation_digest":"sha256:19060fca3135668066c0d5f72d54fa067a40973ab916738472d81154d394c32f","observation_id":"771dd334-511b-4a90-b6e3-134abeed11a4","resolution":{"observed_at":"2026-08-14T05:50:07.200036Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T05:50:07.180134Z","title":"Amant, Chetan Bansal, Victor Ruhle, Anoop Kulkarni, Steve Kofsky, and Saravan Rajmohan","venue":null,"work_id":"a5e7065b-1fc6-4447-8fb0-7d09e251073c","year":2025},"citing_paper":{"arxiv_id":"2608.13499","last_updated":"2026-08-13T17:28:54Z","snapshot_observed_at":"2026-08-15T13:16:28.671545Z","submitted_at":"2026-08-13T17:28:54Z","title":"OpScale: Operator-level Provisioning and Autoscaling for LLM Serving","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-14T05:50:05.720125Z"},"links":{"citing_paper":"/paper/2608.13499"},"observation_digest":"sha256:13c8e00bed357a914f7df98f017791ccce67be6327c8e2054a2dfcdaebf441e7","observation_id":"924bdd43-ab67-4ad1-a3e8-8567fbd67efc","resolution":{"observed_at":"2026-08-14T05:50:07.185431Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T05:50:07.164125Z","title":"Pod-Attention: Unlocking full prefill-decode overlap for faster LLM inference","venue":null,"work_id":"ca177313-6fa9-4d96-93d3-034f96559285","year":2025},"citing_paper":{"arxiv_id":"2608.13499","last_updated":"2026-08-13T17:28:54Z","snapshot_observed_at":"2026-08-15T13:16:28.671545Z","submitted_at":"2026-08-13T17:28:54Z","title":"OpScale: Operator-level Provisioning and Autoscaling for LLM Serving","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-14T05:50:05.724751Z"},"links":{"citing_paper":"/paper/2608.13499"},"observation_digest":"sha256:cd874886ea91bcf4560ddf39812ed2d0a4499b7a28afdd8f724a14b77510ca66","observation_id":"1be6b991-b591-4ecf-9a07-3c83f133207c","resolution":{"observed_at":"2026-08-14T05:50:07.169644Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T05:50:07.147949Z","title":"A simulation analysis of sojourn times in a Jackson network","venue":null,"work_id":"86865103-c5db-4834-9844-d60843afb133","year":1980},"citing_paper":{"arxiv_id":"2608.13499","last_updated":"2026-08-13T17:28:54Z","snapshot_observed_at":"2026-08-15T13:16:28.671545Z","submitted_at":"2026-08-13T17:28:54Z","title":"OpScale: Operator-level Provisioning and Autoscaling for LLM Serving","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-14T05:50:05.729378Z"},"links":{"citing_paper":"/paper/2608.13499"},"observation_digest":"sha256:b6c8617e294fefab630f7c202d652fadd156359dd2d0366db937c9719d569a98","observation_id":"9df997e1-05d3-4d02-b715-96b58486c268","resolution":{"observed_at":"2026-08-14T05:50:07.154096Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T05:50:07.130618Z","title":"Horizontal Pod Autoscaling.http: //kubernetes.io/docs/concepts/workloads/ autoscaling/horizontal-pod-autoscale, 2026","venue":null,"work_id":"428811d5-2282-4a1e-a2f3-9cbd6217e83c","year":2026},"citing_paper":{"arxiv_id":"2608.13499","last_updated":"2026-08-13T17:28:54Z","snapshot_observed_at":"2026-08-15T13:16:28.671545Z","submitted_at":"2026-08-13T17:28:54Z","title":"OpScale: Operator-level Provisioning and Autoscaling for LLM Serving","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-14T05:50:05.733920Z"},"links":{"citing_paper":"/paper/2608.13499"},"observation_digest":"sha256:0070f291691dae70bec1b3e58244bf6af7cb123fd5916361c6dc92c846a5761a","observation_id":"0933a7e9-53cd-46fa-94a9-99b1c2001f57","resolution":{"observed_at":"2026-08-14T05:50:07.136303Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T05:50:07.114329Z","title":"AlpaServe: Statistical multiplexing with model parallelism for deep learning serving","venue":null,"work_id":"ee2ce94e-b35c-42a2-b3be-1c29c9cd6f09","year":2023},"citing_paper":{"arxiv_id":"2608.13499","last_updated":"2026-08-13T17:28:54Z","snapshot_observed_at":"2026-08-15T13:16:28.671545Z","submitted_at":"2026-08-13T17:28:54Z","title":"OpScale: Operator-level Provisioning and Autoscaling for LLM Serving","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-14T05:50:05.738575Z"},"links":{"citing_paper":"/paper/2608.13499"},"observation_digest":"sha256:4fddd0e83c111f3b0d5c1f6c49d1c130915a876c831579fb76adc72eaf2d6b5c","observation_id":"cf63d73f-90b7-4d69-b425-1e235e89558a","resolution":{"observed_at":"2026-08-14T05:50:07.120020Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T05:50:05.743399Z","title":"Bullet: Boosting GPU utilization for LLM serving via dynamic spatial-temporal orchestration.arXiv preprint arXiv:2504.19516, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.13499","last_updated":"2026-08-13T17:28:54Z","snapshot_observed_at":"2026-08-15T13:16:28.671545Z","submitted_at":"2026-08-13T17:28:54Z","title":"OpScale: Operator-level Provisioning and Autoscaling for LLM Serving","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-14T05:50:05.743399Z"},"links":{"citing_paper":"/paper/2608.13499"},"observation_digest":"sha256:f8f51e2ad5c75eecf8b769f329bc78bbb09c1d35514d0ffea999c7befac464b3","observation_id":"40f2fa55-c712-490a-90a2-19b5dfa05470","resolution":{"observed_at":"2026-08-14T05:50:05.743399Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T05:50:05.747912Z","title":"Expert-as-a-service: Towards efficient, scalable, and robust large-scale MoE serving.arXiv preprint arXiv:2509.17863, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.13499","last_updated":"2026-08-13T17:28:54Z","snapshot_observed_at":"2026-08-15T13:16:28.671545Z","submitted_at":"2026-08-13T17:28:54Z","title":"OpScale: Operator-level Provisioning and Autoscaling for LLM Serving","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-14T05:50:05.747912Z"},"links":{"citing_paper":"/paper/2608.13499"},"observation_digest":"sha256:1e582a72df4505d59f5514924b87b6e9f02af8a837ee78b43c047fa06cd749f9","observation_id":"596d5588-e44b-4d1a-aa4e-c4eabcf73a61","resolution":{"observed_at":"2026-08-14T05:50:05.747912Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T05:50:07.098982Z","title":"Azure VM NDm-A100-v4 sizes series.https://learn.microsoft.com/en-us/ azure/virtual-machines/sizes/ gpu-accelerated/ndma100v4-series, 2024","venue":null,"work_id":"9c69ed6c-7c99-4eb0-a081-d012c601d210","year":2024},"citing_paper":{"arxiv_id":"2608.13499","last_updated":"2026-08-13T17:28:54Z","snapshot_observed_at":"2026-08-15T13:16:28.671545Z","submitted_at":"2026-08-13T17:28:54Z","title":"OpScale: Operator-level Provisioning and Autoscaling for LLM Serving","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-14T05:50:05.752956Z"},"links":{"citing_paper":"/paper/2608.13499"},"observation_digest":"sha256:615c6056f3df619f74fd9b0f70e3d756ee548a80fa7dbe67658c217c19e7f4d2","observation_id":"d1dbf675-28a5-47e3-b64b-87848916c88f","resolution":{"observed_at":"2026-08-14T05:50:07.104092Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T05:50:07.083097Z","title":"Azure VM ND GB200-v6 sizes series.https://learn.microsoft.com/en-us/ azure/virtual-machines/sizes/ gpu-accelerated/nd-gb200-v6-series, 2026","venue":null,"work_id":"bbef9384-05a1-44ad-8205-713181bb0396","year":2026},"citing_paper":{"arxiv_id":"2608.13499","last_updated":"2026-08-13T17:28:54Z","snapshot_observed_at":"2026-08-15T13:16:28.671545Z","submitted_at":"2026-08-13T17:28:54Z","title":"OpScale: Operator-level Provisioning and Autoscaling for LLM Serving","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-14T05:50:05.757849Z"},"links":{"citing_paper":"/paper/2608.13499"},"observation_digest":"sha256:f289c56a2d7578fa76df15523ebd0cf4368a2ff1d450156aee5a0310d912e063","observation_id":"efb460e8-7530-4193-b535-370850ecdbc9","resolution":{"observed_at":"2026-08-14T05:50:07.088207Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T05:50:07.064897Z","title":"Documentation on NVIDIA Multi-Instance GPU (MIG).https://www.nvidia.com/en-us/ technologies/multi-instance-gpu/, 2025","venue":null,"work_id":"a382a11a-9437-473f-a720-2eebd75b448c","year":2025},"citing_paper":{"arxiv_id":"2608.13499","last_updated":"2026-08-13T17:28:54Z","snapshot_observed_at":"2026-08-15T13:16:28.671545Z","submitted_at":"2026-08-13T17:28:54Z","title":"OpScale: Operator-level Provisioning and Autoscaling for LLM Serving","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-14T05:50:05.762812Z"},"links":{"citing_paper":"/paper/2608.13499"},"observation_digest":"sha256:654ee0e3ccd68332650563a1b3135e66706df337c9c1a78b5f984f14fa94d725","observation_id":"3641b40c-e348-4ac7-8ca9-790f26fa96d8","resolution":{"observed_at":"2026-08-14T05:50:07.070755Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T05:50:07.047967Z","title":"Documentation on NVIDIA Multi-Process Service (MPS).https: //docs.nvidia.com/deploy/mps/index.html, 2025","venue":null,"work_id":"504112f9-56e1-46c0-84da-46f78f57b9fa","year":2025},"citing_paper":{"arxiv_id":"2608.13499","last_updated":"2026-08-13T17:28:54Z","snapshot_observed_at":"2026-08-15T13:16:28.671545Z","submitted_at":"2026-08-13T17:28:54Z","title":"OpScale: Operator-level Provisioning and Autoscaling for LLM Serving","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-14T05:50:05.767872Z"},"links":{"citing_paper":"/paper/2608.13499"},"observation_digest":"sha256:f648ba3f8e127df5329a82280419ee5ba1b5c9390f17cab6c1d8bba31c1278c6","observation_id":"e86991f8-af73-4fe2-a167-f491c9dcca21","resolution":{"observed_at":"2026-08-14T05:50:07.053891Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T05:50:07.028647Z","title":"Nsight Systems.https: //developer.nvidia.com/nsight-systems, 2025","venue":null,"work_id":"50a25d17-be34-4379-9e66-2a876efb4b9a","year":2025},"citing_paper":{"arxiv_id":"2608.13499","last_updated":"2026-08-13T17:28:54Z","snapshot_observed_at":"2026-08-15T13:16:28.671545Z","submitted_at":"2026-08-13T17:28:54Z","title":"OpScale: Operator-level Provisioning and Autoscaling for LLM Serving","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-14T05:50:05.773012Z"},"links":{"citing_paper":"/paper/2608.13499"},"observation_digest":"sha256:6a687ff0e3dfaa958447a25fc6a4bfbfe318f2b54fdd7ccb4700ca86b5281764","observation_id":"0bd48cb5-f5fc-42b6-8193-4375d847a04f","resolution":{"observed_at":"2026-08-14T05:50:07.034860Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T05:50:07.011571Z","title":"NVIDIA DCGM","venue":null,"work_id":"a294c791-a2a7-4441-bd84-1ca09fdc05b6","year":2025},"citing_paper":{"arxiv_id":"2608.13499","last_updated":"2026-08-13T17:28:54Z","snapshot_observed_at":"2026-08-15T13:16:28.671545Z","submitted_at":"2026-08-13T17:28:54Z","title":"OpScale: Operator-level Provisioning and Autoscaling for LLM Serving","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-14T05:50:05.778394Z"},"links":{"citing_paper":"/paper/2608.13499"},"observation_digest":"sha256:246b6b6d50a4b6459c6b738c5875b09e279bfa68d7459083dbafbad31609b7c8","observation_id":"e116b5e0-47e3-4200-a5de-75bc36638341","resolution":{"observed_at":"2026-08-14T05:50:07.016725Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T05:50:06.996116Z","title":"NVIDIA Green Context Documentation","venue":null,"work_id":"aa3e273a-11f4-4258-8041-d534454035b5","year":2025},"citing_paper":{"arxiv_id":"2608.13499","last_updated":"2026-08-13T17:28:54Z","snapshot_observed_at":"2026-08-15T13:16:28.671545Z","submitted_at":"2026-08-13T17:28:54Z","title":"OpScale: Operator-level Provisioning and Autoscaling for LLM Serving","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-14T05:50:05.783729Z"},"links":{"citing_paper":"/paper/2608.13499"},"observation_digest":"sha256:e24ee57c17b876eef45a005620762b4a660163f66b2ec101547c638b6713db7d","observation_id":"d6a147d4-c168-4e68-9113-0e72f68209e9","resolution":{"observed_at":"2026-08-14T05:50:07.001372Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T05:50:06.980636Z","title":"Introducing ChatGPT","venue":null,"work_id":"9665752a-f1d9-42d2-9d9c-ea08ae1adb78","year":2022},"citing_paper":{"arxiv_id":"2608.13499","last_updated":"2026-08-13T17:28:54Z","snapshot_observed_at":"2026-08-15T13:16:28.671545Z","submitted_at":"2026-08-13T17:28:54Z","title":"OpScale: Operator-level Provisioning and Autoscaling for LLM Serving","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-14T05:50:05.788270Z"},"links":{"citing_paper":"/paper/2608.13499"},"observation_digest":"sha256:1f209f12d5757e606ac9b08bd0991eeae0a366d102a46270169dab40b554e338","observation_id":"8961f5ca-e16c-4bdc-831e-0c389ce7d350","resolution":{"observed_at":"2026-08-14T05:50:06.985505Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T05:50:06.964005Z","title":"ChatGPT Codex","venue":null,"work_id":"8941bc5f-e353-45f2-a6d6-155cb1efb979","year":2025},"citing_paper":{"arxiv_id":"2608.13499","last_updated":"2026-08-13T17:28:54Z","snapshot_observed_at":"2026-08-15T13:16:28.671545Z","submitted_at":"2026-08-13T17:28:54Z","title":"OpScale: Operator-level Provisioning and Autoscaling for LLM Serving","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-14T05:50:05.792789Z"},"links":{"citing_paper":"/paper/2608.13499"},"observation_digest":"sha256:6e3f194ff95ada9d2695766a61da76da6b1ffae1a7b8ad6a04aecbe36718fe5c","observation_id":"224920e1-81ed-4248-a34b-74b856d9128d","resolution":{"observed_at":"2026-08-14T05:50:06.969116Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T05:50:06.945471Z","title":"Introducing Deep Research","venue":null,"work_id":"ac1b2e69-3289-438a-a6f7-d2593f1c3d9b","year":2025},"citing_paper":{"arxiv_id":"2608.13499","last_updated":"2026-08-13T17:28:54Z","snapshot_observed_at":"2026-08-15T13:16:28.671545Z","submitted_at":"2026-08-13T17:28:54Z","title":"OpScale: Operator-level Provisioning and Autoscaling for LLM Serving","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-14T05:50:05.797378Z"},"links":{"citing_paper":"/paper/2608.13499"},"observation_digest":"sha256:f9a87fa6c26e82d26dfdab47fd3c0806bb46f7403af42720f7c34943cebb28eb","observation_id":"baacd22e-c7a1-4486-845d-de686a0bc8be","resolution":{"observed_at":"2026-08-14T05:50:06.953208Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2512.04123","last_updated":"2026-06-04T19:57:38Z","snapshot_observed_at":"2026-08-08T18:53:52.266559Z","submitted_at":"2025-12-02T16:45:10Z","title":"Measuring Agents in Production","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2512.04123","snapshot_observed_at":"2026-08-14T05:50:05.802315Z","title":"Measuring agents in production.arXiv preprint arXiv:2512.04123, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.13499","last_updated":"2026-08-13T17:28:54Z","snapshot_observed_at":"2026-08-15T13:16:28.671545Z","submitted_at":"2026-08-13T17:28:54Z","title":"OpScale: Operator-level Provisioning and Autoscaling for LLM Serving","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-14T05:50:05.802315Z"},"links":{"cited_paper":"/paper/2512.04123","citing_paper":"/paper/2608.13499"},"observation_digest":"sha256:8dd9020f6bbc2229acdb94df63604292ce9520e80a560da2a0a7351f90745d1e","observation_id":"3837b475-2097-41fa-ae2e-c81018b4ab17","resolution":{"observed_at":"2026-08-14T05:50:05.802315Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T05:50:06.927060Z","title":"Splitwise: Efficient generative LLM inference using phase splitting","venue":null,"work_id":"a81d6655-b6f7-4b68-a434-bfec3a1835d2","year":2024},"citing_paper":{"arxiv_id":"2608.13499","last_updated":"2026-08-13T17:28:54Z","snapshot_observed_at":"2026-08-15T13:16:28.671545Z","submitted_at":"2026-08-13T17:28:54Z","title":"OpScale: Operator-level Provisioning and Autoscaling for LLM Serving","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-14T05:50:05.807259Z"},"links":{"citing_paper":"/paper/2608.13499"},"observation_digest":"sha256:6bf075f687eb5977e7c10781ded0adcc273086f7d89c6cf19b68f8cd2d458645","observation_id":"8b9ac719-4b3d-4f6a-b323-812a45d49b10","resolution":{"observed_at":"2026-08-14T05:50:06.933862Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.08090","last_updated":"2025-01-14T12:57:40Z","snapshot_observed_at":"2026-08-14T10:59:44.643354Z","submitted_at":"2025-01-14T12:57:40Z","title":"Hierarchical Autoscaling for Large Language Model Serving with Chiron","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.08090","snapshot_observed_at":"2026-08-14T05:50:05.811741Z","title":"Hierarchical autoscaling for large language model serving with Chiron.arXiv preprint arXiv:2501.08090, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.13499","last_updated":"2026-08-13T17:28:54Z","snapshot_observed_at":"2026-08-15T13:16:28.671545Z","submitted_at":"2026-08-13T17:28:54Z","title":"OpScale: Operator-level Provisioning and Autoscaling for LLM Serving","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-14T05:50:05.811741Z"},"links":{"cited_paper":"/paper/2501.08090","citing_paper":"/paper/2608.13499"},"observation_digest":"sha256:e11cd49f397cd0488d4c5592ee0e85dee72e2686f6af073f068626cf9a29354a","observation_id":"2fbffbea-b1bd-4ee0-9093-4e8370d75c5d","resolution":{"observed_at":"2026-08-14T05:50:05.811741Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T05:50:06.907895Z","title":"Gonzalez, Ion Stoica, and Harry Xu","venue":null,"work_id":"d3846588-e9ef-43d0-8f48-b940b37100d4","year":2025},"citing_paper":{"arxiv_id":"2608.13499","last_updated":"2026-08-13T17:28:54Z","snapshot_observed_at":"2026-08-15T13:16:28.671545Z","submitted_at":"2026-08-13T17:28:54Z","title":"OpScale: Operator-level Provisioning and Autoscaling for LLM Serving","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-14T05:50:05.816995Z"},"links":{"citing_paper":"/paper/2608.13499"},"observation_digest":"sha256:964919059ee5b0cb939211b7b8543991fac46ecbcc5320c907fdf58e44906830","observation_id":"3341333c-74d4-42ea-a23c-6251c0123749","resolution":{"observed_at":"2026-08-14T05:50:06.913621Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.00079","last_updated":"2025-09-03T14:56:29Z","snapshot_observed_at":"2026-08-12T23:36:32.131457Z","submitted_at":"2024-06-24T02:05:32Z","title":"Mooncake: A KVCache-centric Disaggregated Architecture for LLM Serving","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.00079","snapshot_observed_at":"2026-08-14T05:50:05.822059Z","title":"Mooncake: A KVCache-centric Disaggregated Architecture for LLM Serving.arXiv preprint arXiv:2407.00079, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.13499","last_updated":"2026-08-13T17:28:54Z","snapshot_observed_at":"2026-08-15T13:16:28.671545Z","submitted_at":"2026-08-13T17:28:54Z","title":"OpScale: Operator-level Provisioning and Autoscaling for LLM Serving","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-14T05:50:05.822059Z"},"links":{"cited_paper":"/paper/2407.00079","citing_paper":"/paper/2608.13499"},"observation_digest":"sha256:e7588edbe2ed1a80534b20b04192a46a698243092e501a74bb5d1bd9b6b722e1","observation_id":"d3859c9b-3093-4662-b8c0-68de115c3106","resolution":{"observed_at":"2026-08-14T05:50:05.822059Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T05:50:06.891414Z","title":"FIRM: An intelligent fine-grained resource management framework for SLO-oriented microservices","venue":null,"work_id":"3c2f953d-ad12-4136-b1eb-7b7ba643baf7","year":2020},"citing_paper":{"arxiv_id":"2608.13499","last_updated":"2026-08-13T17:28:54Z","snapshot_observed_at":"2026-08-15T13:16:28.671545Z","submitted_at":"2026-08-13T17:28:54Z","title":"OpScale: Operator-level Provisioning and Autoscaling for LLM Serving","version":1},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-14T05:50:05.827601Z"},"links":{"citing_paper":"/paper/2608.13499"},"observation_digest":"sha256:53fe9eb1d5cd4c6983c8f6779ec6ebe923d150cc9f509f7b507d51c0a27967d3","observation_id":"6bbe73cc-c469-4f6d-908c-af50bd5574b0","resolution":{"observed_at":"2026-08-14T05:50:06.896442Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T05:50:05.832604Z","title":"ModServe: Scalable and resource-efficient large multimodal model serving.arXiv preprint arXiv:2502.00937, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.13499","last_updated":"2026-08-13T17:28:54Z","snapshot_observed_at":"2026-08-15T13:16:28.671545Z","submitted_at":"2026-08-13T17:28:54Z","title":"OpScale: Operator-level Provisioning and Autoscaling for LLM Serving","version":1},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-14T05:50:05.832604Z"},"links":{"citing_paper":"/paper/2608.13499"},"observation_digest":"sha256:1d16cc31225a22075744f5207250c9e0b54d9fd59b05dd7eb39e9354ecba8d2f","observation_id":"c5fb8028-6a9e-41e4-89fd-e945f150bfc9","resolution":{"observed_at":"2026-08-14T05:50:05.832604Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T05:50:06.876156Z","title":"Power-aware deep learning model serving with µ-Serve","venue":null,"work_id":"8bf569d0-8a44-44dc-a9d3-6ab55a5cea75","year":2024},"citing_paper":{"arxiv_id":"2608.13499","last_updated":"2026-08-13T17:28:54Z","snapshot_observed_at":"2026-08-15T13:16:28.671545Z","submitted_at":"2026-08-13T17:28:54Z","title":"OpScale: Operator-level Provisioning and Autoscaling for LLM Serving","version":1},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-08-14T05:50:05.837628Z"},"links":{"citing_paper":"/paper/2608.13499"},"observation_digest":"sha256:a97c17aef3009fced8663e1671cb4e38c595111bcdfbbe0ca06ff99c6d0c58c8","observation_id":"e409514e-9414-42e2-a094-11fa45ee5b8b","resolution":{"observed_at":"2026-08-14T05:50:06.881336Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T05:50:06.860311Z","title":"USHER: Holistic interference avoidance for resource optimized ML inference","venue":null,"work_id":"e50d158d-8fdb-46ec-a8a1-b9637a5ad67c","year":2024},"citing_paper":{"arxiv_id":"2608.13499","last_updated":"2026-08-13T17:28:54Z","snapshot_observed_at":"2026-08-15T13:16:28.671545Z","submitted_at":"2026-08-13T17:28:54Z","title":"OpScale: Operator-level Provisioning and Autoscaling for LLM Serving","version":1},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-14T05:50:05.842612Z"},"links":{"citing_paper":"/paper/2608.13499"},"observation_digest":"sha256:1393a1b497255db7a2b0fbe9d570600ba44d5cb6b6f4df172fd3e52a500670ee","observation_id":"dc9df5f0-ac90-482d-aef0-696a9fcb3403","resolution":{"observed_at":"2026-08-14T05:50:06.865672Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.05460","last_updated":"2025-06-28T03:53:17Z","snapshot_observed_at":"2026-08-15T09:45:11.396132Z","submitted_at":"2024-12-25T10:11:31Z","title":"Efficiently Serving Large Multimodal Models Using EPD Disaggregation","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.05460","snapshot_observed_at":"2026-08-14T05:50:05.847560Z","title":"Efficiently serving large multimedia models using EPD disaggregation.arXiv preprint arXiv:2501.05460, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.13499","last_updated":"2026-08-13T17:28:54Z","snapshot_observed_at":"2026-08-15T13:16:28.671545Z","submitted_at":"2026-08-13T17:28:54Z","title":"OpScale: Operator-level Provisioning and Autoscaling for LLM Serving","version":1},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-08-14T05:50:05.847560Z"},"links":{"cited_paper":"/paper/2501.05460","citing_paper":"/paper/2608.13499"},"observation_digest":"sha256:ce576d759e3182c2fe01f0d4650f5679ac16890c2a4cb8e3b9ddf387cb75b17b","observation_id":"c9ad68a2-064b-453d-9a35-568e673923f5","resolution":{"observed_at":"2026-08-14T05:50:05.847560Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T05:50:06.843274Z","title":"DynamoLLM: Designing LLM inference clusters for performance and energy efficiency","venue":null,"work_id":"f76514cb-002a-4080-a284-179652a7f1e7","year":2025},"citing_paper":{"arxiv_id":"2608.13499","last_updated":"2026-08-13T17:28:54Z","snapshot_observed_at":"2026-08-15T13:16:28.671545Z","submitted_at":"2026-08-13T17:28:54Z","title":"OpScale: Operator-level Provisioning and Autoscaling for LLM Serving","version":1},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-08-14T05:50:05.852309Z"},"links":{"citing_paper":"/paper/2608.13499"},"observation_digest":"sha256:e2d1435de3cef1814e91c927c8ae038ec0e47dc81684247408ea286c9787aadc","observation_id":"6c846edf-bf77-4054-80c6-90ea870afdf4","resolution":{"observed_at":"2026-08-14T05:50:06.849384Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T05:50:06.825424Z","title":"Orion: Interference-aware, fine-grained GPU sharing for ML applications","venue":null,"work_id":"23e7c73c-18cb-4398-93d0-5e5c652787ee","year":2024},"citing_paper":{"arxiv_id":"2608.13499","last_updated":"2026-08-13T17:28:54Z","snapshot_observed_at":"2026-08-15T13:16:28.671545Z","submitted_at":"2026-08-13T17:28:54Z","title":"OpScale: Operator-level Provisioning and Autoscaling for LLM Serving","version":1},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-08-14T05:50:05.857081Z"},"links":{"citing_paper":"/paper/2608.13499"},"observation_digest":"sha256:a5eea2605f85ad41c289a67a5414de9358aa93ad85708840dd9e9d6691eabb0f","observation_id":"434b7763-0b9d-44ff-bb37-b8b3c97e6e35","resolution":{"observed_at":"2026-08-14T05:50:06.831719Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.03648","last_updated":"2025-02-22T07:07:38Z","snapshot_observed_at":"2026-08-15T07:42:34.001187Z","submitted_at":"2025-02-22T07:07:38Z","title":"AIBrix: Towards Scalable, Cost-Effective Large Language Model Inference Infrastructure","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.03648","snapshot_observed_at":"2026-08-14T05:50:05.862038Z","title":"AIBrix: Towards scalable, cost-effective large language model inference infrastructure.arXiv preprint arXiv:2504.03648, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.13499","last_updated":"2026-08-13T17:28:54Z","snapshot_observed_at":"2026-08-15T13:16:28.671545Z","submitted_at":"2026-08-13T17:28:54Z","title":"OpScale: Operator-level Provisioning and Autoscaling for LLM Serving","version":1},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-08-14T05:50:05.862038Z"},"links":{"cited_paper":"/paper/2504.03648","citing_paper":"/paper/2608.13499"},"observation_digest":"sha256:91b49e6878571a5a6ab50bf4393e4c17535a16a6562cee1438442e0fbc2a8139","observation_id":"005c611f-6144-43ca-b66f-cb6d1b47f03d","resolution":{"observed_at":"2026-08-14T05:50:05.862038Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T05:50:06.809064Z","title":"Distributed Inference and Serving","venue":null,"work_id":"8a799302-0e63-4f42-819d-0b2e61a02eb2","year":2024},"citing_paper":{"arxiv_id":"2608.13499","last_updated":"2026-08-13T17:28:54Z","snapshot_observed_at":"2026-08-15T13:16:28.671545Z","submitted_at":"2026-08-13T17:28:54Z","title":"OpScale: Operator-level Provisioning and Autoscaling for LLM Serving","version":1},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-08-14T05:50:05.866937Z"},"links":{"citing_paper":"/paper/2608.13499"},"observation_digest":"sha256:7f77ea38342d0860ef7248b83d859fd28ebcf0a73b50633f6c5153c63d5e4eff","observation_id":"31a91211-922b-4548-aba9-dd83a56be30a","resolution":{"observed_at":"2026-08-14T05:50:06.814067Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T05:50:06.791054Z","title":"vLLM Profiler.https://docs.vllm.ai/en/ stable/contributing/profiling/, 2025","venue":null,"work_id":"9f1a2a46-38c6-431a-866f-b3b9b9dd5323","year":2025},"citing_paper":{"arxiv_id":"2608.13499","last_updated":"2026-08-13T17:28:54Z","snapshot_observed_at":"2026-08-15T13:16:28.671545Z","submitted_at":"2026-08-13T17:28:54Z","title":"OpScale: Operator-level Provisioning and Autoscaling for LLM Serving","version":1},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-08-14T05:50:05.873942Z"},"links":{"citing_paper":"/paper/2608.13499"},"observation_digest":"sha256:795e343ab0ae12fce4ad7cfdf33b60a1f329dab27713c30a7f1f2a9c92b2833c","observation_id":"e08d7657-36b4-4a3c-8e1a-2631e020b52e","resolution":{"observed_at":"2026-08-14T05:50:06.797238Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T05:50:05.879083Z","title":"Step-3 is large yet affordable: Model-system co-design for cost-effective decoding.arXiv preprint arXiv:2507.19427, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.13499","last_updated":"2026-08-13T17:28:54Z","snapshot_observed_at":"2026-08-15T13:16:28.671545Z","submitted_at":"2026-08-13T17:28:54Z","title":"OpScale: Operator-level Provisioning and Autoscaling for LLM Serving","version":1},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-08-14T05:50:05.879083Z"},"links":{"citing_paper":"/paper/2608.13499"},"observation_digest":"sha256:c8f316cabcd14799ceb53fa0ede03d7083328a915b419fd320005aabd85100a4","observation_id":"3f41d75b-28c2-43c8-920c-4d854305b720","resolution":{"observed_at":"2026-08-14T05:50:05.879083Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T05:50:06.772975Z","title":"Autothrottle: A practical bi-level approach to resource management for SLO-targeted microservices","venue":null,"work_id":"30ccfb31-c23f-4e3a-b7d3-c34be8f33d0e","year":2024},"citing_paper":{"arxiv_id":"2608.13499","last_updated":"2026-08-13T17:28:54Z","snapshot_observed_at":"2026-08-15T13:16:28.671545Z","submitted_at":"2026-08-13T17:28:54Z","title":"OpScale: Operator-level Provisioning and Autoscaling for LLM Serving","version":1},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-08-14T05:50:05.884034Z"},"links":{"citing_paper":"/paper/2608.13499"},"observation_digest":"sha256:709ab1891e7215354454bf3ac247bea1149281472506cbeb089be3f991722a18","observation_id":"7cf61f99-b599-4973-bf5f-447bf464003c","resolution":{"observed_at":"2026-08-14T05:50:06.778567Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T05:50:06.756078Z","title":"DeepScaling: microservices autoscaling for stable cpu utilization in large scale cloud systems","venue":null,"work_id":"8b763453-bdb7-4b3b-aa0d-be1f63ef39c1","year":2022},"citing_paper":{"arxiv_id":"2608.13499","last_updated":"2026-08-13T17:28:54Z","snapshot_observed_at":"2026-08-15T13:16:28.671545Z","submitted_at":"2026-08-13T17:28:54Z","title":"OpScale: Operator-level Provisioning and Autoscaling for LLM Serving","version":1},"reference_index":64,"source":"pdf_text","source_observed_at":"2026-08-14T05:50:05.889694Z"},"links":{"citing_paper":"/paper/2608.13499"},"observation_digest":"sha256:9998dc2dfadba58c67f42a728723fcb2e5eaa09fbe2433324a9d10331ceaa71c","observation_id":"c200162e-ce8a-440c-86a6-587522659815","resolution":{"observed_at":"2026-08-14T05:50:06.761726Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T05:50:06.740392Z","title":"Aegaeon: Effective GPU pooling for concurrent LLM serving on the market","venue":null,"work_id":"382d3d2f-c1a5-4950-981d-2436e744a129","year":2025},"citing_paper":{"arxiv_id":"2608.13499","last_updated":"2026-08-13T17:28:54Z","snapshot_observed_at":"2026-08-15T13:16:28.671545Z","submitted_at":"2026-08-13T17:28:54Z","title":"OpScale: Operator-level Provisioning and Autoscaling for LLM Serving","version":1},"reference_index":65,"source":"pdf_text","source_observed_at":"2026-08-14T05:50:05.894643Z"},"links":{"citing_paper":"/paper/2608.13499"},"observation_digest":"sha256:dfa22c9e71b3746a14da739179284a51dc4ada4c097d3555f58cfe52de2ddd61","observation_id":"623bbe61-556d-4498-be9c-999e8dba87dc","resolution":{"observed_at":"2026-08-14T05:50:06.745531Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2508.08448","last_updated":"2025-08-11T20:11:43Z","snapshot_observed_at":"2026-08-14T04:30:44.940455Z","submitted_at":"2025-08-11T20:11:43Z","title":"Towards Efficient and Practical GPU Multitasking in the Era of LLM","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2508.08448","snapshot_observed_at":"2026-08-14T05:50:05.899552Z","title":"Towards efficient and practical GPU multitasking in the era of LLM.arXiv preprint arXiv:2508.08448, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.13499","last_updated":"2026-08-13T17:28:54Z","snapshot_observed_at":"2026-08-15T13:16:28.671545Z","submitted_at":"2026-08-13T17:28:54Z","title":"OpScale: Operator-level Provisioning and Autoscaling for LLM Serving","version":1},"reference_index":66,"source":"pdf_text","source_observed_at":"2026-08-14T05:50:05.899552Z"},"links":{"cited_paper":"/paper/2508.08448","citing_paper":"/paper/2608.13499"},"observation_digest":"sha256:9acd3abec8a647ed4f2bfdbd9fe1fae20368aebb03aa27af627ef8b586067da0","observation_id":"b6bf6a03-1957-45cf-af65-9246a48b9bd8","resolution":{"observed_at":"2026-08-14T05:50:05.899552Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2505.04021","last_updated":"2026-06-10T23:04:53Z","snapshot_observed_at":"2026-08-13T18:02:25.530101Z","submitted_at":"2025-05-06T23:38:33Z","title":"Prism: Cost-Efficient Multi-LLM Serving via GPU Memory Ballooning","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.04021","snapshot_observed_at":"2026-08-14T05:50:05.905030Z","title":"Prism: Unleashing GPU sharing for cost-efficient multi-LLM serving.arXiv preprint arXiv:2505.04021, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.13499","last_updated":"2026-08-13T17:28:54Z","snapshot_observed_at":"2026-08-15T13:16:28.671545Z","submitted_at":"2026-08-13T17:28:54Z","title":"OpScale: Operator-level Provisioning and Autoscaling for LLM Serving","version":1},"reference_index":67,"source":"pdf_text","source_observed_at":"2026-08-14T05:50:05.905030Z"},"links":{"cited_paper":"/paper/2505.04021","citing_paper":"/paper/2608.13499"},"observation_digest":"sha256:85e020f5646c4109359a1b54620711199de3a535d2b7907ba3bd9e097157fa7a","observation_id":"d7174262-3b1c-40b6-9d86-119ebe8cdb6e","resolution":{"observed_at":"2026-08-14T05:50:05.905030Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T05:50:06.724604Z","title":"DistServe: Disaggregating prefill and decoding for goodput-optimized large language model serving","venue":null,"work_id":"da39d5ae-f8c9-4efc-ab09-fa621cf5242a","year":2024},"citing_paper":{"arxiv_id":"2608.13499","last_updated":"2026-08-13T17:28:54Z","snapshot_observed_at":"2026-08-15T13:16:28.671545Z","submitted_at":"2026-08-13T17:28:54Z","title":"OpScale: Operator-level Provisioning and Autoscaling for LLM Serving","version":1},"reference_index":68,"source":"pdf_text","source_observed_at":"2026-08-14T05:50:05.910551Z"},"links":{"citing_paper":"/paper/2608.13499"},"observation_digest":"sha256:d8de107c74d1688e4034898bb15165733580d6039a8b4cafbbcb47c38072a755","observation_id":"c956474c-8c6b-4762-b6d5-434fa09e7650","resolution":{"observed_at":"2026-08-14T05:50:06.730071Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-14T05:50:06.706265Z","title":"NanoFlow: Towards optimal large language model serving throughput","venue":null,"work_id":"b8a3dbd9-a833-4279-b7e0-c67e73bc6641","year":2025},"citing_paper":{"arxiv_id":"2608.13499","last_updated":"2026-08-13T17:28:54Z","snapshot_observed_at":"2026-08-15T13:16:28.671545Z","submitted_at":"2026-08-13T17:28:54Z","title":"OpScale: Operator-level Provisioning and Autoscaling for LLM Serving","version":1},"reference_index":69,"source":"pdf_text","source_observed_at":"2026-08-14T05:50:05.915430Z"},"links":{"citing_paper":"/paper/2608.13499"},"observation_digest":"sha256:55b88baf435a9e5cce7647bdbb69c1259bfe783b1c058530cc73cc8a12d10fbc","observation_id":"e268cb21-855f-4aa2-b214-de1acc1dcedf","resolution":{"observed_at":"2026-08-14T05:50:06.713390Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-15T06:32:42.880941+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.02263","last_updated":"2025-07-26T15:29:10Z","snapshot_observed_at":"2026-08-08T01:33:26.740871Z","submitted_at":"2025-04-03T04:20:44Z","title":"MegaScale-Infer: Serving Mixture-of-Experts at Scale with Disaggregated Expert Parallelism","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.02263","snapshot_observed_at":"2026-08-14T05:50:05.920154Z","title":"MegaScale-Infer: Serving mixture-of-experts at scale with disaggregated expert parallelism.arXiv preprint arXiv:2504.02263, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.13499","last_updated":"2026-08-13T17:28:54Z","snapshot_observed_at":"2026-08-15T13:16:28.671545Z","submitted_at":"2026-08-13T17:28:54Z","title":"OpScale: Operator-level Provisioning and Autoscaling for LLM Serving","version":1},"reference_index":70,"source":"pdf_text","source_observed_at":"2026-08-14T05:50:05.920154Z"},"links":{"cited_paper":"/paper/2504.02263","citing_paper":"/paper/2608.13499"},"observation_digest":"sha256:28778084f9d5aec9e5c429f450620d410877e3900a5d41bc65b753aabf3534e1","observation_id":"6fd1a60d-da8f-4b10-9a6f-d793d5a53838","resolution":{"observed_at":"2026-08-14T05:50:05.920154Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2506.12708","last_updated":"2025-06-19T12:27:10Z","snapshot_observed_at":"2026-08-14T05:02:32.366760Z","submitted_at":"2025-06-15T03:41:34Z","title":"Serving Large Language Models on Huawei CloudMatrix384","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2506.12708","snapshot_observed_at":"2026-08-14T05:50:05.925477Z","title":"Serving large language models on Huawei CloudMatrix384.arXiv preprint arXiv:2506.12708, 2025","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2608.13499","last_updated":"2026-08-13T17:28:54Z","snapshot_observed_at":"2026-08-15T13:16:28.671545Z","submitted_at":"2026-08-13T17:28:54Z","title":"OpScale: Operator-level Provisioning and Autoscaling for LLM Serving","version":1},"reference_index":71,"source":"pdf_text","source_observed_at":"2026-08-14T05:50:05.925477Z"},"links":{"cited_paper":"/paper/2506.12708","citing_paper":"/paper/2608.13499"},"observation_digest":"sha256:ce1079de1f26c74acdf50da45118ff883b3ac974d6a28c8dd0691f8a4ee1a4dc","observation_id":"3e3af8a0-484b-4851-98c2-04187c6ab46c","resolution":{"observed_at":"2026-08-14T05:50:05.925477Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2608.13499","last_updated":"2026-08-13T17:28:54Z","latest_version":1,"primary_category":"cs.DC","snapshot_observed_at":"2026-08-15T13:16:28.671545Z","submitted_at":"2026-08-13T17:28:54Z","title":"OpScale: Operator-level Provisioning and Autoscaling for LLM Serving"},"reference_resolution":{"displayed":71,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":19,"verified_exact":0,"verified_fuzzy":52},"total_outbound_references":71},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-15T06:32:42.880941+00:00","source":"crossref"},{"observed_at":"2026-08-15T06:32:39.529945+00:00","source":"retraction_watch"}],"thesis":"As of 15 August 2026, this Paper Citation Record lists 71 of 71 outbound references and 0 inbound Pith citation observations for arXiv:2608.13499."}