{"as_of":"2026-08-10T17:48:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:9ace679bd50e081aea4273cacb8520b527089fdb2f91ce4a649e17d5573fb8d9","coverage":[{"denominator":47,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":47,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-06T10:22:42.600721Z","state":"measured"},{"denominator":47,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":47,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-10T06:31:04.303077+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2508.00234/citation-record","integrity":"/paper/2508.00234/integrity","json":"/paper/2508.00234/citation-record.json","paper":"/paper/2508.00234"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T10:22:43.086141Z","title":"AIoT smart home via autonomous LLM agents,","venue":null,"work_id":"68cfa256-a397-483f-9845-5bfc47e49666","year":2024},"citing_paper":{"arxiv_id":"2508.00234","last_updated":"2025-08-01T00:45:15Z","snapshot_observed_at":"2026-08-10T03:38:13.939485Z","submitted_at":"2025-08-01T00:45:15Z","title":"Quality-of-Service Aware LLM Routing for Edge Computing with Multiple Experts","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-06T10:22:42.438667Z"},"links":{"citing_paper":"/paper/2508.00234"},"observation_digest":"sha256:3867857dc759ebcd0f01bad6ec36516c1c4e76f8772c041a823ccbc5f27bb48b","observation_id":"a7efcac1-e101-4a8f-aee0-b89606f8e866","resolution":{"observed_at":"2026-08-06T10:22:43.090189Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T10:22:43.076641Z","title":"Large language models for human- ai co-creation of robotic dance performances,","venue":null,"work_id":"81a65eab-2532-4343-9cdf-34ce2d5195ef","year":2024},"citing_paper":{"arxiv_id":"2508.00234","last_updated":"2025-08-01T00:45:15Z","snapshot_observed_at":"2026-08-10T03:38:13.939485Z","submitted_at":"2025-08-01T00:45:15Z","title":"Quality-of-Service Aware LLM Routing for Edge Computing with Multiple Experts","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-06T10:22:42.442311Z"},"links":{"citing_paper":"/paper/2508.00234"},"observation_digest":"sha256:1e8cdac92df651f46d3c989e25630361bed5c515da8a5c3b74587d04d473cae4","observation_id":"5becd7f5-7a18-4a0b-8b99-0a7ff6399765","resolution":{"observed_at":"2026-08-06T10:22:43.079977Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T10:22:43.067436Z","title":"EdgeFM: Leveraging foundation model for open-set learning on the edge,","venue":null,"work_id":"76bb55fb-b706-4e5d-b27c-ff5af0303f9d","year":2023},"citing_paper":{"arxiv_id":"2508.00234","last_updated":"2025-08-01T00:45:15Z","snapshot_observed_at":"2026-08-10T03:38:13.939485Z","submitted_at":"2025-08-01T00:45:15Z","title":"Quality-of-Service Aware LLM Routing for Edge Computing with Multiple Experts","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-06T10:22:42.445752Z"},"links":{"citing_paper":"/paper/2508.00234"},"observation_digest":"sha256:c4d22870b6644eb2176aa0a2b20894d6f737b277b217631ce2a7d1f97a3628c5","observation_id":"66562e3f-b0ec-43cb-95a3-6d327ed826de","resolution":{"observed_at":"2026-08-06T10:22:43.070683Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.03131","last_updated":"2024-05-06T02:55:50Z","snapshot_observed_at":"2026-08-08T16:31:34.860721Z","submitted_at":"2024-05-06T02:55:50Z","title":"WDMoE: Wireless Distributed Large Language Models with Mixture of Experts","version":1},"cited_work":{"arxiv_id":"2405.03131","doi":null,"metadata_source":"pith","pith_arxiv_id":"2405.03131","snapshot_observed_at":"2026-08-06T10:22:42.845110Z","title":"WDMoE: Wireless Distributed Large Language Models with Mixture of Experts","venue":"cs.IT","work_id":"ade63cc5-ad46-4c69-9c63-ce8468b51727","year":2024},"citing_paper":{"arxiv_id":"2508.00234","last_updated":"2025-08-01T00:45:15Z","snapshot_observed_at":"2026-08-10T03:38:13.939485Z","submitted_at":"2025-08-01T00:45:15Z","title":"Quality-of-Service Aware LLM Routing for Edge Computing with Multiple Experts","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-06T10:22:42.449489Z"},"links":{"cited_paper":"/paper/2405.03131","citing_paper":"/paper/2508.00234"},"observation_digest":"sha256:a9a77439eb8a6e291fc49cb5df8949cb087ef3f67c87d3bc6392b12945e4b5a2","observation_id":"fa1a07e2-b807-4ec9-afdd-780f71e7630e","resolution":{"observed_at":"2026-08-06T10:22:42.849405Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.05156","last_updated":"2024-03-14T14:17:57Z","snapshot_observed_at":"2026-07-30T18:58:06.958154Z","submitted_at":"2024-03-08T08:47:48Z","title":"On Protecting the Data Privacy of Large Language Models (LLMs): A Survey","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.05156","snapshot_observed_at":"2026-08-06T10:22:42.453414Z","title":"On protecting the data privacy of large language models (LLMs): A survey,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.00234","last_updated":"2025-08-01T00:45:15Z","snapshot_observed_at":"2026-08-10T03:38:13.939485Z","submitted_at":"2025-08-01T00:45:15Z","title":"Quality-of-Service Aware LLM Routing for Edge Computing with Multiple Experts","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-06T10:22:42.453414Z"},"links":{"cited_paper":"/paper/2403.05156","citing_paper":"/paper/2508.00234"},"observation_digest":"sha256:34dc7eb775b301923db51817307833ed1f3698eb85627b9a7809b6bcecb038c0","observation_id":"e29751e0-7f8d-4110-a95a-ee4b27e27192","resolution":{"observed_at":"2026-08-06T10:22:42.453414Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T10:22:43.058451Z","title":"Edge intelligence: Paving the last mile of artificial intelligence with edge computing,","venue":null,"work_id":"1dab9184-a7b1-4eb0-b5c4-388d4092b3c1","year":2019},"citing_paper":{"arxiv_id":"2508.00234","last_updated":"2025-08-01T00:45:15Z","snapshot_observed_at":"2026-08-10T03:38:13.939485Z","submitted_at":"2025-08-01T00:45:15Z","title":"Quality-of-Service Aware LLM Routing for Edge Computing with Multiple Experts","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-06T10:22:42.457777Z"},"links":{"citing_paper":"/paper/2508.00234"},"observation_digest":"sha256:95e664c249350a28b2173b47b0ba04cbc3bdfb8bf9e4d9e76e85a91842d9e4f5","observation_id":"94d46d84-12c1-4b01-9fc2-9714dd1b58c2","resolution":{"observed_at":"2026-08-06T10:22:43.061734Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2301.03220","last_updated":"2023-01-09T09:30:23Z","snapshot_observed_at":"2026-08-07T06:40:34.051221Z","submitted_at":"2023-01-09T09:30:23Z","title":"Enabling AI-Generated Content (AIGC) Services in Wireless Edge Networks","version":1},"cited_work":{"arxiv_id":"2301.03220","doi":null,"metadata_source":"pith","pith_arxiv_id":"2301.03220","snapshot_observed_at":"2026-08-06T10:22:42.823164Z","title":"Enabling AI-Generated Content (AIGC) Services in Wireless Edge Networks","venue":"cs.AI","work_id":"a587d834-31ef-4fc2-be8e-396922027a19","year":2023},"citing_paper":{"arxiv_id":"2508.00234","last_updated":"2025-08-01T00:45:15Z","snapshot_observed_at":"2026-08-10T03:38:13.939485Z","submitted_at":"2025-08-01T00:45:15Z","title":"Quality-of-Service Aware LLM Routing for Edge Computing with Multiple Experts","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-06T10:22:42.461675Z"},"links":{"cited_paper":"/paper/2301.03220","citing_paper":"/paper/2508.00234"},"observation_digest":"sha256:24aa55a8b77c94f7feae784d3ffb2af0580a911c2b5e3a2efa7bf4f878741902","observation_id":"057fe7f9-4c30-4db0-bcf3-867bc48a5d07","resolution":{"observed_at":"2026-08-06T10:22:42.826912Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2402.06942","last_updated":"2024-02-10T13:00:37Z","snapshot_observed_at":"2026-08-07T06:40:40.754311Z","submitted_at":"2024-02-10T13:00:37Z","title":"Toward Scalable Generative AI via Mixture of Experts in Mobile Edge Networks","version":1},"cited_work":{"arxiv_id":"2402.06942","doi":null,"metadata_source":"pith","pith_arxiv_id":"2402.06942","snapshot_observed_at":"2026-08-06T10:22:42.808494Z","title":"Toward Scalable Generative AI via Mixture of Experts in Mobile Edge Networks","venue":"cs.NI","work_id":"3c5b35fc-9678-4e91-a701-183276a4b7ee","year":2024},"citing_paper":{"arxiv_id":"2508.00234","last_updated":"2025-08-01T00:45:15Z","snapshot_observed_at":"2026-08-10T03:38:13.939485Z","submitted_at":"2025-08-01T00:45:15Z","title":"Quality-of-Service Aware LLM Routing for Edge Computing with Multiple Experts","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-06T10:22:42.465930Z"},"links":{"cited_paper":"/paper/2402.06942","citing_paper":"/paper/2508.00234"},"observation_digest":"sha256:28a2f23419b7a050c06f1214e9e156905dc2350088c47c0a47a98798ba13bb51","observation_id":"178e19ba-07ac-4d24-ae51-acba3c2955d2","resolution":{"observed_at":"2026-08-06T10:22:42.812607Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T10:22:43.048717Z","title":"LLM-Blender: Ensembling large language models with pairwise ranking and generative fusion,","venue":null,"work_id":"7433d02a-dd90-4612-b2be-339c8953ff9a","year":2023},"citing_paper":{"arxiv_id":"2508.00234","last_updated":"2025-08-01T00:45:15Z","snapshot_observed_at":"2026-08-10T03:38:13.939485Z","submitted_at":"2025-08-01T00:45:15Z","title":"Quality-of-Service Aware LLM Routing for Edge Computing with Multiple Experts","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-06T10:22:42.469627Z"},"links":{"citing_paper":"/paper/2508.00234"},"observation_digest":"sha256:5d885d3c66d566505c1a2cbdcc3e7b618b639689385261b2fceb8be235456bfc","observation_id":"a15f1f24-6d08-4389-8ebc-a4dcd44239c4","resolution":{"observed_at":"2026-08-06T10:22:43.052200Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.13510","last_updated":"2025-01-07T18:16:17Z","snapshot_observed_at":"2026-07-06T19:05:26.961192Z","submitted_at":"2024-08-24T08:12:22Z","title":"Intelligent Router for LLM Workloads: Improving Performance Through Workload-Aware Load Balancing","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.13510","snapshot_observed_at":"2026-08-06T10:22:42.473633Z","title":"Intelligent router for LLM workloads: Improving performance through workload-aware schedul- ing,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.00234","last_updated":"2025-08-01T00:45:15Z","snapshot_observed_at":"2026-08-10T03:38:13.939485Z","submitted_at":"2025-08-01T00:45:15Z","title":"Quality-of-Service Aware LLM Routing for Edge Computing with Multiple Experts","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-06T10:22:42.473633Z"},"links":{"cited_paper":"/paper/2408.13510","citing_paper":"/paper/2508.00234"},"observation_digest":"sha256:0bb62f803778528247e68880609de77192b3de9c2d291ff7955c70660e10f441","observation_id":"c1324dbe-b74d-4485-a844-e9374342ebe8","resolution":{"observed_at":"2026-08-06T10:22:42.473633Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T10:22:43.039204Z","title":"Orca: A distributed serving system for Transformer-based generative models,","venue":null,"work_id":"990944bb-2caa-4d0c-8200-619914631185","year":2022},"citing_paper":{"arxiv_id":"2508.00234","last_updated":"2025-08-01T00:45:15Z","snapshot_observed_at":"2026-08-10T03:38:13.939485Z","submitted_at":"2025-08-01T00:45:15Z","title":"Quality-of-Service Aware LLM Routing for Edge Computing with Multiple Experts","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-06T10:22:42.477410Z"},"links":{"citing_paper":"/paper/2508.00234"},"observation_digest":"sha256:00bb728fa4ed62b2d105e0207a38ad38a7af97f55c02edb5243ff7623ed05d29","observation_id":"5851814e-fe47-4e49-9f96-335532db538d","resolution":{"observed_at":"2026-08-06T10:22:43.042587Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T10:22:43.028907Z","title":"Efficient memory management for large language model serving with PagedAttention,","venue":null,"work_id":"9a585468-ae5d-42ab-a086-5bb15ccea9f1","year":2023},"citing_paper":{"arxiv_id":"2508.00234","last_updated":"2025-08-01T00:45:15Z","snapshot_observed_at":"2026-08-10T03:38:13.939485Z","submitted_at":"2025-08-01T00:45:15Z","title":"Quality-of-Service Aware LLM Routing for Edge Computing with Multiple Experts","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-06T10:22:42.480957Z"},"links":{"citing_paper":"/paper/2508.00234"},"observation_digest":"sha256:870dd9edc461a582cac43c049421b81720b7c0d79207c1a91b99d7e5cfedc42d","observation_id":"f57748b3-8b08-4739-8117-89915ee233ad","resolution":{"observed_at":"2026-08-06T10:22:43.033027Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.12320","last_updated":"2024-10-23T18:11:42Z","snapshot_observed_at":"2026-07-06T19:04:31.201611Z","submitted_at":"2024-08-22T11:57:07Z","title":"TensorOpera Router: A Multi-Model Router for Efficient LLM Inference","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.12320","snapshot_observed_at":"2026-08-06T10:22:42.484238Z","title":"PolyRouter: A multi-LLM querying system,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.00234","last_updated":"2025-08-01T00:45:15Z","snapshot_observed_at":"2026-08-10T03:38:13.939485Z","submitted_at":"2025-08-01T00:45:15Z","title":"Quality-of-Service Aware LLM Routing for Edge Computing with Multiple Experts","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-06T10:22:42.484238Z"},"links":{"cited_paper":"/paper/2408.12320","citing_paper":"/paper/2508.00234"},"observation_digest":"sha256:07a34f9001314810583e4d3f6a76e9edff1fe98109cae5b344efa47b1509b7eb","observation_id":"d6fb5336-512d-48cd-8483-794d817d97f1","resolution":{"observed_at":"2026-08-06T10:22:42.484238Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.08692","last_updated":"2023-11-15T04:40:43Z","snapshot_observed_at":"2026-07-06T16:47:45.504109Z","submitted_at":"2023-11-15T04:40:43Z","title":"Routing to the Expert: Efficient Reward-guided Ensemble of Large Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.08692","snapshot_observed_at":"2026-08-06T10:22:42.487916Z","title":"Routing to the expert: Efficient reward-guided ensemble of large language models,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2508.00234","last_updated":"2025-08-01T00:45:15Z","snapshot_observed_at":"2026-08-10T03:38:13.939485Z","submitted_at":"2025-08-01T00:45:15Z","title":"Quality-of-Service Aware LLM Routing for Edge Computing with Multiple Experts","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-06T10:22:42.487916Z"},"links":{"cited_paper":"/paper/2311.08692","citing_paper":"/paper/2508.00234"},"observation_digest":"sha256:8fc05f05e6aba567b809687df0febdb0d1b5414eaadf46f7acd009204f3aa84a","observation_id":"c3c4b001-6546-45ba-b1e2-86507aa5b3cd","resolution":{"observed_at":"2026-08-06T10:22:42.487916Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.14618","last_updated":"2024-04-22T23:06:42Z","snapshot_observed_at":"2026-08-08T18:42:25.777499Z","submitted_at":"2024-04-22T23:06:42Z","title":"Hybrid LLM: Cost-Efficient and Quality-Aware Query Routing","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.14618","snapshot_observed_at":"2026-08-06T10:22:42.491345Z","title":"Hybrid LLM: Cost-efficient and quality-aware query routing,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.00234","last_updated":"2025-08-01T00:45:15Z","snapshot_observed_at":"2026-08-10T03:38:13.939485Z","submitted_at":"2025-08-01T00:45:15Z","title":"Quality-of-Service Aware LLM Routing for Edge Computing with Multiple Experts","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-06T10:22:42.491345Z"},"links":{"cited_paper":"/paper/2404.14618","citing_paper":"/paper/2508.00234"},"observation_digest":"sha256:5067ebe4511adb4404fd43483eac812b847c7f05b358b216e5dbf965c8215b6e","observation_id":"5d215f12-f43b-4d2c-be98-bc4d39c890e6","resolution":{"observed_at":"2026-08-06T10:22:42.491345Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.18665","last_updated":"2025-02-23T08:50:33Z","snapshot_observed_at":"2026-07-30T15:06:18.011623Z","submitted_at":"2024-06-26T18:10:22Z","title":"RouteLLM: Learning to Route LLMs with Preference Data","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.18665","snapshot_observed_at":"2026-08-06T10:22:42.495457Z","title":"RouteLLM: Learning to route LLMs with preference data,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.00234","last_updated":"2025-08-01T00:45:15Z","snapshot_observed_at":"2026-08-10T03:38:13.939485Z","submitted_at":"2025-08-01T00:45:15Z","title":"Quality-of-Service Aware LLM Routing for Edge Computing with Multiple Experts","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-06T10:22:42.495457Z"},"links":{"cited_paper":"/paper/2406.18665","citing_paper":"/paper/2508.00234"},"observation_digest":"sha256:af71c209926a2ac11236269a12977a46cfb0690759712550252cc8860a688714","observation_id":"9e1a3c60-1439-493e-8f61-0320821726da","resolution":{"observed_at":"2026-08-06T10:22:42.495457Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.17644","last_updated":"2025-05-26T16:16:43Z","snapshot_observed_at":"2026-08-10T12:43:15.683604Z","submitted_at":"2024-01-31T07:52:48Z","title":"BurstGPT: A Real-world Workload Dataset to Optimize LLM Serving Systems","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.17644","snapshot_observed_at":"2026-08-06T10:22:42.499001Z","title":"Towards efficient and reliable LLM serving: A real- world workload study,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.00234","last_updated":"2025-08-01T00:45:15Z","snapshot_observed_at":"2026-08-10T03:38:13.939485Z","submitted_at":"2025-08-01T00:45:15Z","title":"Quality-of-Service Aware LLM Routing for Edge Computing with Multiple Experts","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-06T10:22:42.499001Z"},"links":{"cited_paper":"/paper/2401.17644","citing_paper":"/paper/2508.00234"},"observation_digest":"sha256:1142690fd07cb8119eb2efa6018b884fc481d5cc8f902c8b5beba5bc59657947","observation_id":"c8f02fcc-0387-4b39-9431-998ed626a0a2","resolution":{"observed_at":"2026-08-06T10:22:42.499001Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.08509","last_updated":"2024-11-25T17:35:07Z","snapshot_observed_at":"2026-08-09T09:17:59.811981Z","submitted_at":"2024-04-12T14:46:15Z","title":"Efficient Interactive LLM Serving with Proxy Model-based Sequence Length Prediction","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.08509","snapshot_observed_at":"2026-08-06T10:22:42.502527Z","title":"Efficient interactive LLM serving with proxy model-based sequence length prediction,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.00234","last_updated":"2025-08-01T00:45:15Z","snapshot_observed_at":"2026-08-10T03:38:13.939485Z","submitted_at":"2025-08-01T00:45:15Z","title":"Quality-of-Service Aware LLM Routing for Edge Computing with Multiple Experts","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-06T10:22:42.502527Z"},"links":{"cited_paper":"/paper/2404.08509","citing_paper":"/paper/2508.00234"},"observation_digest":"sha256:514f63e64cddfc5cc3e07a1883b5f8e1caacc34dc69affdf627a65cb2575d06b","observation_id":"6f8399dc-ff72-4234-bc1e-d302dff8fba4","resolution":{"observed_at":"2026-08-06T10:22:42.502527Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T10:22:43.018561Z","title":"S3: Increasing gpu utiliza- tion during generative inference for higher throughput,","venue":null,"work_id":"6e013f42-64d6-47c9-bee5-63043d821116","year":2023},"citing_paper":{"arxiv_id":"2508.00234","last_updated":"2025-08-01T00:45:15Z","snapshot_observed_at":"2026-08-10T03:38:13.939485Z","submitted_at":"2025-08-01T00:45:15Z","title":"Quality-of-Service Aware LLM Routing for Edge Computing with Multiple Experts","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-06T10:22:42.506125Z"},"links":{"citing_paper":"/paper/2508.00234"},"observation_digest":"sha256:c8fcf669155b1e2f6fe2d1fa1ff451c00a23cbf77d6539d71f4ace83e5b66647","observation_id":"5feb3bae-8f4e-4fc5-a454-b3d2bd10255d","resolution":{"observed_at":"2026-08-06T10:22:43.022410Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T10:22:43.008858Z","title":"FlexGen: High-throughput generative inference of large language models with a single gpu,","venue":null,"work_id":"cb506243-5e23-462e-9b56-895ab55c9fcf","year":2023},"citing_paper":{"arxiv_id":"2508.00234","last_updated":"2025-08-01T00:45:15Z","snapshot_observed_at":"2026-08-10T03:38:13.939485Z","submitted_at":"2025-08-01T00:45:15Z","title":"Quality-of-Service Aware LLM Routing for Edge Computing with Multiple Experts","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-06T10:22:42.509785Z"},"links":{"citing_paper":"/paper/2508.00234"},"observation_digest":"sha256:9585bbe9f50e3e40faf3256201fda8ff756e5b75d1bad9d476e7e0ddf4213ff2","observation_id":"ff93db18-1f56-45f4-af19-a093d6a6f59e","resolution":{"observed_at":"2026-08-06T10:22:43.012284Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T10:22:42.998752Z","title":"FlashAttention: Fast and memory-efficient exact attention with io-awareness,","venue":null,"work_id":"805bec3e-7be4-4fe8-8145-a3c55c25ec70","year":2022},"citing_paper":{"arxiv_id":"2508.00234","last_updated":"2025-08-01T00:45:15Z","snapshot_observed_at":"2026-08-10T03:38:13.939485Z","submitted_at":"2025-08-01T00:45:15Z","title":"Quality-of-Service Aware LLM Routing for Edge Computing with Multiple Experts","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-06T10:22:42.512936Z"},"links":{"citing_paper":"/paper/2508.00234"},"observation_digest":"sha256:dcddfce160b822b4f8127e56c450277a61b80fce3b0014d908838dd515abf5b1","observation_id":"31825be3-8091-4d3c-942e-620e5bc4d5ab","resolution":{"observed_at":"2026-08-06T10:22:43.002330Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T10:22:42.989106Z","title":"HeteGen: Efficient heterogeneous parallel inference for large language models on resource-constrained devices,","venue":null,"work_id":"61799329-a1f7-4d94-bfba-04ba7dad4782","year":2024},"citing_paper":{"arxiv_id":"2508.00234","last_updated":"2025-08-01T00:45:15Z","snapshot_observed_at":"2026-08-10T03:38:13.939485Z","submitted_at":"2025-08-01T00:45:15Z","title":"Quality-of-Service Aware LLM Routing for Edge Computing with Multiple Experts","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-06T10:22:42.516096Z"},"links":{"citing_paper":"/paper/2508.00234"},"observation_digest":"sha256:1467fdf910ee6306de4154acaf16baece84272f407a6fd8f9cbaeabfc9ba619f","observation_id":"8e83768f-be94-4e1b-9b39-1bfced14ec38","resolution":{"observed_at":"2026-08-06T10:22:42.992601Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T10:22:42.979155Z","title":"ExeGPT: Constraint-aware resource scheduling for LLM inference,","venue":null,"work_id":"cf0fd711-0fb9-4659-bede-ef4e041041c0","year":2024},"citing_paper":{"arxiv_id":"2508.00234","last_updated":"2025-08-01T00:45:15Z","snapshot_observed_at":"2026-08-10T03:38:13.939485Z","submitted_at":"2025-08-01T00:45:15Z","title":"Quality-of-Service Aware LLM Routing for Edge Computing with Multiple Experts","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-06T10:22:42.519086Z"},"links":{"citing_paper":"/paper/2508.00234"},"observation_digest":"sha256:c5f8fe9b610512177c9417b5e41a9ef4d9462b3b4e3f3dfc27571df8a57151b2","observation_id":"6a1aaa9b-af2a-45c9-952d-ed89d8481d74","resolution":{"observed_at":"2026-08-06T10:22:42.983278Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.19708","last_updated":"2024-06-30T23:50:38Z","snapshot_observed_at":"2026-08-03T11:15:07.121206Z","submitted_at":"2024-03-23T10:42:49Z","title":"Cost-Efficient Large Language Model Serving for Multi-turn Conversations with CachedAttention","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.19708","snapshot_observed_at":"2026-08-06T10:22:42.522976Z","title":"AttentionStore: Cost-effective attention reuse across multi- turn conversations in large language model serving,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.00234","last_updated":"2025-08-01T00:45:15Z","snapshot_observed_at":"2026-08-10T03:38:13.939485Z","submitted_at":"2025-08-01T00:45:15Z","title":"Quality-of-Service Aware LLM Routing for Edge Computing with Multiple Experts","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-06T10:22:42.522976Z"},"links":{"cited_paper":"/paper/2403.19708","citing_paper":"/paper/2508.00234"},"observation_digest":"sha256:c9253c4fcce7658c0409255aa4c3b43130cb806d3021c686acd84d5699025dfa","observation_id":"120312c5-8335-48a5-8248-6cb87de67a0f","resolution":{"observed_at":"2026-08-06T10:22:42.522976Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T10:22:42.968508Z","title":"Splitwise: Efficient generative LLM inference using phase splitting,","venue":null,"work_id":"b6acab46-7004-433f-b30e-086c18ed4e09","year":2024},"citing_paper":{"arxiv_id":"2508.00234","last_updated":"2025-08-01T00:45:15Z","snapshot_observed_at":"2026-08-10T03:38:13.939485Z","submitted_at":"2025-08-01T00:45:15Z","title":"Quality-of-Service Aware LLM Routing for Edge Computing with Multiple Experts","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-06T10:22:42.526269Z"},"links":{"citing_paper":"/paper/2508.00234"},"observation_digest":"sha256:9a38d9b2b7fbab408576f211d8ac2ea9f830d7d431b0a66ee7af080d0b1f0924","observation_id":"ed792d59-abd2-4839-bdd1-2093adadd75e","resolution":{"observed_at":"2026-08-06T10:22:42.971972Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.09670","last_updated":"2024-06-06T15:50:51Z","snapshot_observed_at":"2026-08-05T00:28:57.370523Z","submitted_at":"2024-01-18T01:03:38Z","title":"DistServe: Disaggregating Prefill and Decoding for Goodput-optimized Large Language Model Serving","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.09670","snapshot_observed_at":"2026-08-06T10:22:42.529491Z","title":"DistServe: Disaggregating prefill and decoding for goodput-optimized large language model serving,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.00234","last_updated":"2025-08-01T00:45:15Z","snapshot_observed_at":"2026-08-10T03:38:13.939485Z","submitted_at":"2025-08-01T00:45:15Z","title":"Quality-of-Service Aware LLM Routing for Edge Computing with Multiple Experts","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-06T10:22:42.529491Z"},"links":{"cited_paper":"/paper/2401.09670","citing_paper":"/paper/2508.00234"},"observation_digest":"sha256:0e94da7fd3f33c1bfe568fd071c0416940e79ef03aa96e78caa250045a702a83","observation_id":"99811022-c3ff-47f0-adb3-784c8973e78f","resolution":{"observed_at":"2026-08-06T10:22:42.529491Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.06089","last_updated":"2024-07-08T16:29:08Z","snapshot_observed_at":"2026-08-10T07:13:23.169291Z","submitted_at":"2024-07-08T16:29:08Z","title":"Merge, Ensemble, and Cooperate! A Survey on Collaborative Strategies in the Era of Large Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.06089","snapshot_observed_at":"2026-08-06T10:22:42.532900Z","title":"Merge, ensemble, and cooperate! a survey on collaborative strategies in the era of large language models,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.00234","last_updated":"2025-08-01T00:45:15Z","snapshot_observed_at":"2026-08-10T03:38:13.939485Z","submitted_at":"2025-08-01T00:45:15Z","title":"Quality-of-Service Aware LLM Routing for Edge Computing with Multiple Experts","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-06T10:22:42.532900Z"},"links":{"cited_paper":"/paper/2407.06089","citing_paper":"/paper/2508.00234"},"observation_digest":"sha256:5ea9d5535c4ba97879d8d9b8ccde4ba683ffed3ba92a51f3e244c1516d08a24a","observation_id":"29ca24c1-7612-4688-bc65-4af530a53ced","resolution":{"observed_at":"2026-08-06T10:22:42.532900Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2309.15789","last_updated":"2023-09-27T17:08:40Z","snapshot_observed_at":"2026-08-06T08:57:12.331150Z","submitted_at":"2023-09-27T17:08:40Z","title":"Large Language Model Routing with Benchmark Datasets","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.15789","snapshot_observed_at":"2026-08-06T10:22:42.536535Z","title":"Large language model routing with benchmark datasets,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2508.00234","last_updated":"2025-08-01T00:45:15Z","snapshot_observed_at":"2026-08-10T03:38:13.939485Z","submitted_at":"2025-08-01T00:45:15Z","title":"Quality-of-Service Aware LLM Routing for Edge Computing with Multiple Experts","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-06T10:22:42.536535Z"},"links":{"cited_paper":"/paper/2309.15789","citing_paper":"/paper/2508.00234"},"observation_digest":"sha256:bbeb01ef6626813421f2c313e7aac26bfbe7946144dc785bd1b7aebf7105a70c","observation_id":"972eceb0-f50d-4179-990e-02817071252e","resolution":{"observed_at":"2026-08-06T10:22:42.536535Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.19296","last_updated":"2026-07-20T02:16:43Z","snapshot_observed_at":"2026-08-06T07:29:39.723831Z","submitted_at":"2024-04-30T06:55:45Z","title":"Octopus v4: Graph of language models","version":2},"cited_work":{"arxiv_id":"2404.19296","doi":null,"metadata_source":"pith","pith_arxiv_id":"2404.19296","snapshot_observed_at":"2026-08-06T10:22:42.694271Z","title":"Octopus v4: Graph of language models","venue":"cs.CL","work_id":"3550046b-1b32-4649-82b8-2001550376ca","year":2024},"citing_paper":{"arxiv_id":"2508.00234","last_updated":"2025-08-01T00:45:15Z","snapshot_observed_at":"2026-08-10T03:38:13.939485Z","submitted_at":"2025-08-01T00:45:15Z","title":"Quality-of-Service Aware LLM Routing for Edge Computing with Multiple Experts","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-06T10:22:42.540098Z"},"links":{"cited_paper":"/paper/2404.19296","citing_paper":"/paper/2508.00234"},"observation_digest":"sha256:98c80074b191374f544ba9289831cb9c409e60b96a159abb6b78b2105b7d5e67","observation_id":"907191d6-53e4-470b-a88a-964cba56d57b","resolution":{"observed_at":"2026-08-06T10:22:42.699685Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.03834","last_updated":"2025-03-17T15:08:47Z","snapshot_observed_at":"2026-08-05T19:53:12.719014Z","submitted_at":"2024-10-04T18:02:48Z","title":"GraphRouter: A Graph-based Router for LLM Selections","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.03834","snapshot_observed_at":"2026-08-06T10:22:42.543528Z","title":"GraphRouter: A graph-based router for LLM selections,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.00234","last_updated":"2025-08-01T00:45:15Z","snapshot_observed_at":"2026-08-10T03:38:13.939485Z","submitted_at":"2025-08-01T00:45:15Z","title":"Quality-of-Service Aware LLM Routing for Edge Computing with Multiple Experts","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-06T10:22:42.543528Z"},"links":{"cited_paper":"/paper/2410.03834","citing_paper":"/paper/2508.00234"},"observation_digest":"sha256:eea7dfa1cfe23d571ffbb6e2e3d8f0e2f691e52872326eebad1b395c8f8a37b5","observation_id":"959d65b7-5550-4a95-9b9c-bbbd9e40ae62","resolution":{"observed_at":"2026-08-06T10:22:42.543528Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2409.15518","last_updated":"2024-10-29T00:20:45Z","snapshot_observed_at":"2026-08-06T08:17:19.907980Z","submitted_at":"2024-09-23T20:10:10Z","title":"Eagle: Efficient Training-Free Router for Multi-LLM Inference","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.15518","snapshot_observed_at":"2026-08-06T10:22:42.547078Z","title":"Eagle: Efficient training-free router for multi-LLM inference,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.00234","last_updated":"2025-08-01T00:45:15Z","snapshot_observed_at":"2026-08-10T03:38:13.939485Z","submitted_at":"2025-08-01T00:45:15Z","title":"Quality-of-Service Aware LLM Routing for Edge Computing with Multiple Experts","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-06T10:22:42.547078Z"},"links":{"cited_paper":"/paper/2409.15518","citing_paper":"/paper/2508.00234"},"observation_digest":"sha256:04b24ce5611828b7958add10edba7699799722093f14ccc700283155c58845ff","observation_id":"6f7900a0-9cbd-40a5-9bfd-a4e84da2a1da","resolution":{"observed_at":"2026-08-06T10:22:42.547078Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.12031","last_updated":"2024-03-28T17:56:28Z","snapshot_observed_at":"2026-08-09T22:21:33.340392Z","submitted_at":"2024-03-18T17:59:04Z","title":"RouterBench: A Benchmark for Multi-LLM Routing System","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.12031","snapshot_observed_at":"2026-08-06T10:22:42.551092Z","title":"RouterBench: A benchmark for multi- LLM routing system,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.00234","last_updated":"2025-08-01T00:45:15Z","snapshot_observed_at":"2026-08-10T03:38:13.939485Z","submitted_at":"2025-08-01T00:45:15Z","title":"Quality-of-Service Aware LLM Routing for Edge Computing with Multiple Experts","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-06T10:22:42.551092Z"},"links":{"cited_paper":"/paper/2403.12031","citing_paper":"/paper/2508.00234"},"observation_digest":"sha256:0f1dcda2f729e5654f929741d82827d2b07887ad9a5f9306e85639b0c2fcc612","observation_id":"f118e507-38d5-4f06-9c12-fb9cc61baa17","resolution":{"observed_at":"2026-08-06T10:22:42.551092Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T10:22:42.959690Z","title":"Reinforcement learning in dynamic task scheduling: A review,","venue":null,"work_id":"239a6e06-ab5b-4aad-8740-cd4652116061","year":2020},"citing_paper":{"arxiv_id":"2508.00234","last_updated":"2025-08-01T00:45:15Z","snapshot_observed_at":"2026-08-10T03:38:13.939485Z","submitted_at":"2025-08-01T00:45:15Z","title":"Quality-of-Service Aware LLM Routing for Edge Computing with Multiple Experts","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-06T10:22:42.554729Z"},"links":{"citing_paper":"/paper/2508.00234"},"observation_digest":"sha256:16f4da90946ec21952e39bd3562b658b49b111f03d97e455d651fd3642216a22","observation_id":"00649949-c493-4f18-b74b-9b6853a6aac3","resolution":{"observed_at":"2026-08-06T10:22:42.962721Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T10:22:42.949828Z","title":"Collaborative learning-based scheduling for kubernetes-oriented edge-cloud network,","venue":null,"work_id":"7deb2bb7-e4d8-4168-983a-a53ad103247f","year":2023},"citing_paper":{"arxiv_id":"2508.00234","last_updated":"2025-08-01T00:45:15Z","snapshot_observed_at":"2026-08-10T03:38:13.939485Z","submitted_at":"2025-08-01T00:45:15Z","title":"Quality-of-Service Aware LLM Routing for Edge Computing with Multiple Experts","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-06T10:22:42.558272Z"},"links":{"citing_paper":"/paper/2508.00234"},"observation_digest":"sha256:cd78323576f9e96248537264066c2f4366bf046341aa392af624c85e8949f0ac","observation_id":"8ec0fde6-11b0-46de-aa00-7d53d3a016aa","resolution":{"observed_at":"2026-08-06T10:22:42.953572Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T10:22:42.940205Z","title":"Clipper: A low-latency online prediction serving system,","venue":null,"work_id":"222e284e-ff96-40ae-ad03-e83d7759e2d5","year":2017},"citing_paper":{"arxiv_id":"2508.00234","last_updated":"2025-08-01T00:45:15Z","snapshot_observed_at":"2026-08-10T03:38:13.939485Z","submitted_at":"2025-08-01T00:45:15Z","title":"Quality-of-Service Aware LLM Routing for Edge Computing with Multiple Experts","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-06T10:22:42.561384Z"},"links":{"citing_paper":"/paper/2508.00234"},"observation_digest":"sha256:256e4f9029c170d138d803804d37674b118c690175ed3b6ab4d307263b885805","observation_id":"8f2438b1-a10b-460d-9795-efc97d7fd20e","resolution":{"observed_at":"2026-08-06T10:22:42.943669Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T10:22:42.929612Z","title":"Tapfinger: Task place- ment and fine-grained resource allocation for edge machine learning,","venue":null,"work_id":"118dca99-0be5-475f-9468-dd1a72de925c","year":2023},"citing_paper":{"arxiv_id":"2508.00234","last_updated":"2025-08-01T00:45:15Z","snapshot_observed_at":"2026-08-10T03:38:13.939485Z","submitted_at":"2025-08-01T00:45:15Z","title":"Quality-of-Service Aware LLM Routing for Edge Computing with Multiple Experts","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-06T10:22:42.564545Z"},"links":{"citing_paper":"/paper/2508.00234"},"observation_digest":"sha256:17ea34880a72332e511b1d7baec0e498250662621c289998b262d01ad9ebe1aa","observation_id":"4fc7aa00-d5ce-4947-bc46-6a2ca3ac062c","resolution":{"observed_at":"2026-08-06T10:22:42.933924Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T10:22:42.919944Z","title":"The non- stochastic multiarmed bandit problem,","venue":null,"work_id":"921775de-ec7b-4271-913e-ae97f69bfbdb","year":2002},"citing_paper":{"arxiv_id":"2508.00234","last_updated":"2025-08-01T00:45:15Z","snapshot_observed_at":"2026-08-10T03:38:13.939485Z","submitted_at":"2025-08-01T00:45:15Z","title":"Quality-of-Service Aware LLM Routing for Edge Computing with Multiple Experts","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-06T10:22:42.567752Z"},"links":{"citing_paper":"/paper/2508.00234"},"observation_digest":"sha256:f335dbbef9b301588e6c78e5f2c467c6ca07035e664eaa3690fe77d71c80d424","observation_id":"805502f2-61a6-4900-91c7-df0b40ef9c12","resolution":{"observed_at":"2026-08-06T10:22:42.923358Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T10:22:42.910245Z","title":"Alpaca: A strong, replicable instruction- following model,","venue":null,"work_id":"b21c12ec-2fc9-4fa9-9fb5-2c1e4c1dac7f","year":2023},"citing_paper":{"arxiv_id":"2508.00234","last_updated":"2025-08-01T00:45:15Z","snapshot_observed_at":"2026-08-10T03:38:13.939485Z","submitted_at":"2025-08-01T00:45:15Z","title":"Quality-of-Service Aware LLM Routing for Edge Computing with Multiple Experts","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-06T10:22:42.570832Z"},"links":{"citing_paper":"/paper/2508.00234"},"observation_digest":"sha256:b8be7689e5c7dbd097233b0b63890cb0c5521e8bad329c2611226b4a307bb0eb","observation_id":"aeecd7cb-b573-4c6d-86db-3b8f5be55375","resolution":{"observed_at":"2026-08-06T10:22:42.913571Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.12793","last_updated":"2024-07-30T03:58:11Z","snapshot_observed_at":"2026-08-07T13:56:34.167869Z","submitted_at":"2024-06-18T16:58:21Z","title":"ChatGLM: A Family of Large Language Models from GLM-130B to GLM-4 All Tools","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.12793","snapshot_observed_at":"2026-08-06T10:22:42.573898Z","title":"ChatGLM: A family of large language models from GLM-130B to GLM-4 all tools,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.00234","last_updated":"2025-08-01T00:45:15Z","snapshot_observed_at":"2026-08-10T03:38:13.939485Z","submitted_at":"2025-08-01T00:45:15Z","title":"Quality-of-Service Aware LLM Routing for Edge Computing with Multiple Experts","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-06T10:22:42.573898Z"},"links":{"cited_paper":"/paper/2406.12793","citing_paper":"/paper/2508.00234"},"observation_digest":"sha256:c05f271f58974db6e46f2c6b03a5291b82b96fe52abcf9334f0c4e1a673c3450","observation_id":"5ee694a6-7663-4955-b1ba-ba5df4dc6475","resolution":{"observed_at":"2026-08-06T10:22:42.573898Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T10:22:42.899555Z","title":"Introducing Mpt-7b: A new standard for open-source, commercially usable LLMs, 2023,","venue":null,"work_id":"0d9cfed8-6cac-47f9-9bdc-8e856326ec34","year":2023},"citing_paper":{"arxiv_id":"2508.00234","last_updated":"2025-08-01T00:45:15Z","snapshot_observed_at":"2026-08-10T03:38:13.939485Z","submitted_at":"2025-08-01T00:45:15Z","title":"Quality-of-Service Aware LLM Routing for Edge Computing with Multiple Experts","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-06T10:22:42.577353Z"},"links":{"citing_paper":"/paper/2508.00234"},"observation_digest":"sha256:7786f9bdf774a34f9b63dcf0c40e7617a00f39090e454539b71d23c059c59565","observation_id":"e6446bf0-638f-4d86-8d3b-14769f76bdcb","resolution":{"observed_at":"2026-08-06T10:22:42.903359Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T10:22:42.889429Z","title":"BERTScore: Evaluating text generation with BERT,","venue":null,"work_id":"39e028b5-c382-445e-819a-93124d839a96","year":2020},"citing_paper":{"arxiv_id":"2508.00234","last_updated":"2025-08-01T00:45:15Z","snapshot_observed_at":"2026-08-10T03:38:13.939485Z","submitted_at":"2025-08-01T00:45:15Z","title":"Quality-of-Service Aware LLM Routing for Edge Computing with Multiple Experts","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-06T10:22:42.580526Z"},"links":{"citing_paper":"/paper/2508.00234"},"observation_digest":"sha256:7fd482d3ad935eccf7f186f3c6f40bc048a6d22697f3678f129e2b566d8ba73d","observation_id":"ccc560dd-3b4a-4490-ae35-6b60998217b6","resolution":{"observed_at":"2026-08-06T10:22:42.893031Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T10:22:42.878891Z","title":"Soft Actor-Critic: Off- policy maximum entropy deep reinforcement learning with a stochastic actor,","venue":null,"work_id":"c7d16de5-61e6-4f0c-a203-c4c8e11c8b1d","year":2018},"citing_paper":{"arxiv_id":"2508.00234","last_updated":"2025-08-01T00:45:15Z","snapshot_observed_at":"2026-08-10T03:38:13.939485Z","submitted_at":"2025-08-01T00:45:15Z","title":"Quality-of-Service Aware LLM Routing for Edge Computing with Multiple Experts","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-06T10:22:42.583787Z"},"links":{"citing_paper":"/paper/2508.00234"},"observation_digest":"sha256:3cdc8954f99c19b5e174d979b6c2cb4cf502e212d6cb560d2fdc09b68f69bc7b","observation_id":"06130ea9-117f-42d2-958d-ebcc97f7adf5","resolution":{"observed_at":"2026-08-06T10:22:42.882580Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1910.01108","last_updated":"2020-03-01T02:57:50Z","snapshot_observed_at":"2026-08-07T19:07:36.327251Z","submitted_at":"2019-10-02T17:56:28Z","title":"DistilBERT, a distilled version of BERT: smaller, faster, cheaper and lighter","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1910.01108","snapshot_observed_at":"2026-08-06T10:22:42.586945Z","title":"DistilBERT, a distilled version of BERT: Smaller, faster, cheaper and lighter,","venue":null,"work_id":null,"year":1910},"citing_paper":{"arxiv_id":"2508.00234","last_updated":"2025-08-01T00:45:15Z","snapshot_observed_at":"2026-08-10T03:38:13.939485Z","submitted_at":"2025-08-01T00:45:15Z","title":"Quality-of-Service Aware LLM Routing for Edge Computing with Multiple Experts","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-06T10:22:42.586945Z"},"links":{"cited_paper":"/paper/1910.01108","citing_paper":"/paper/2508.00234"},"observation_digest":"sha256:b6b26fd34c09013cd17b53024475926d0cbf114626df00692a9acd6ddd46c8e9","observation_id":"218b2815-e419-46b1-8d47-d6dd83cffc48","resolution":{"observed_at":"2026-08-06T10:22:42.586945Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T10:22:42.867862Z","title":"PyTorch: An im- perative style, high-performance deep learning library,","venue":null,"work_id":"318214a7-b147-411a-9af3-d43e22124ac5","year":2019},"citing_paper":{"arxiv_id":"2508.00234","last_updated":"2025-08-01T00:45:15Z","snapshot_observed_at":"2026-08-10T03:38:13.939485Z","submitted_at":"2025-08-01T00:45:15Z","title":"Quality-of-Service Aware LLM Routing for Edge Computing with Multiple Experts","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-06T10:22:42.590788Z"},"links":{"citing_paper":"/paper/2508.00234"},"observation_digest":"sha256:7159b6887af72b8238dd655d530c1543858717bac2d477b69ef39d4ed83bdab1","observation_id":"349b3954-1ab8-4e79-93c3-68053e355c59","resolution":{"observed_at":"2026-08-06T10:22:42.872062Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.00577","last_updated":"2023-11-27T15:57:06Z","snapshot_observed_at":"2026-08-09T16:10:30.032242Z","submitted_at":"2023-06-01T11:45:45Z","title":"TorchRL: A data-driven decision-making library for PyTorch","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.00577","snapshot_observed_at":"2026-08-06T10:22:42.593937Z","title":"TorchRL: A data-driven decision- making library for PyTorch,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2508.00234","last_updated":"2025-08-01T00:45:15Z","snapshot_observed_at":"2026-08-10T03:38:13.939485Z","submitted_at":"2025-08-01T00:45:15Z","title":"Quality-of-Service Aware LLM Routing for Edge Computing with Multiple Experts","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-06T10:22:42.593937Z"},"links":{"cited_paper":"/paper/2306.00577","citing_paper":"/paper/2508.00234"},"observation_digest":"sha256:1149f9e01a7df180238068641c1d6e836b19fda4772dc16b6722db2b57116c50","observation_id":"c3e0348b-b300-4653-b8be-43de4843abe8","resolution":{"observed_at":"2026-08-06T10:22:42.593937Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1903.02428","last_updated":"2019-04-25T10:06:09Z","snapshot_observed_at":"2026-08-02T18:54:43.326912Z","submitted_at":"2019-03-06T14:50:02Z","title":"Fast Graph Representation Learning with PyTorch Geometric","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1903.02428","snapshot_observed_at":"2026-08-06T10:22:42.597363Z","title":"Fast graph representation learning with PyTorch Geometric,","venue":null,"work_id":null,"year":1903},"citing_paper":{"arxiv_id":"2508.00234","last_updated":"2025-08-01T00:45:15Z","snapshot_observed_at":"2026-08-10T03:38:13.939485Z","submitted_at":"2025-08-01T00:45:15Z","title":"Quality-of-Service Aware LLM Routing for Edge Computing with Multiple Experts","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-06T10:22:42.597363Z"},"links":{"cited_paper":"/paper/1903.02428","citing_paper":"/paper/2508.00234"},"observation_digest":"sha256:6d0962d04e76139d184d81f9476c448776e4b9c34214782253ff3f563ddf275d","observation_id":"744176ad-faf2-4bd9-8275-e68d4e776812","resolution":{"observed_at":"2026-08-06T10:22:42.597363Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T10:22:42.857349Z","title":"BERT: Pre- training of deep bidirectional transformers for language understanding,","venue":null,"work_id":"d73d4db5-9f02-44fd-a05f-edf2c9a397c7","year":2019},"citing_paper":{"arxiv_id":"2508.00234","last_updated":"2025-08-01T00:45:15Z","snapshot_observed_at":"2026-08-10T03:38:13.939485Z","submitted_at":"2025-08-01T00:45:15Z","title":"Quality-of-Service Aware LLM Routing for Edge Computing with Multiple Experts","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-06T10:22:42.600721Z"},"links":{"citing_paper":"/paper/2508.00234"},"observation_digest":"sha256:55ddce0c9ce67fd8d440efe9095ee03718ddb86bb151355f5029ecff2fcadc12","observation_id":"9f7890c3-0601-4b7c-935e-fca2d8f40366","resolution":{"observed_at":"2026-08-06T10:22:42.860805Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2508.00234","last_updated":"2025-08-01T00:45:15Z","latest_version":1,"primary_category":"cs.NI","snapshot_observed_at":"2026-08-10T03:38:13.939485Z","submitted_at":"2025-08-01T00:45:15Z","title":"Quality-of-Service Aware LLM Routing for Edge Computing with Multiple Experts"},"reference_resolution":{"displayed":47,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":19,"verified_exact":4,"verified_fuzzy":24},"total_outbound_references":47},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"thesis":"As of 10 August 2026, this Paper Citation Record lists 47 of 47 outbound references and 0 inbound Pith citation observations for arXiv:2508.00234."}