{"as_of":"2026-08-18T09:44:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:4089486d7f9478a66019dcdca655017ea82d21a7d659a01b5cc1d65c3da11476","coverage":[{"denominator":55,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":55,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-16T10:28:52.545392Z","state":"measured"},{"denominator":57,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":57,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-18T06:34:40.430872+00:00","state":"measured"},{"denominator":2,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":2,"source":"paper_references, paper_reference_links","source_observed_at":"2026-06-27T23:49:28.318260Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-02T15:27:06.035370Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2504.18154","last_updated":"2025-04-25T08:06:22Z","snapshot_observed_at":"2026-08-18T08:38:54.349344Z","submitted_at":"2025-04-25T08:06:22Z","title":"EcoServe: Enabling Cost-effective LLM Serving with Proactive Intra- and Inter-Instance Orchestration","version":1},"cited_work":{"arxiv_id":"2504.18154","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2504.18154","snapshot_observed_at":"2026-07-02T15:27:06.035370Z","title":"arXiv preprint arXiv:2504.18154 , year=","venue":null,"work_id":"9194ca84-f774-433b-8d5f-7cd9f1074f2f","year":2025},"citing_paper":{"arxiv_id":"2605.02189","last_updated":"2026-05-04T03:37:40Z","snapshot_observed_at":"2026-08-12T14:41:49.916145Z","submitted_at":"2026-05-04T03:37:40Z","title":"PipeMax: Enhancing Offline LLM Inference on Commodity GPU Servers","version":1},"reference_index":11,"source":"arxiv_source","source_observed_at":"2026-05-08T18:49:56.357400Z"},"links":{"cited_paper":"/paper/2504.18154","citing_paper":"/paper/2605.02189"},"observation_digest":"sha256:fef1b16b250dd76026b1d16164e8b914c9c17ad7fc549eeb7e6a9b989a00c527","observation_id":"9b6e3136-9875-49ef-be5f-9f31cf4f7e5a","resolution":{"observed_at":"2026-05-09T06:10:41.552912Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.18154","last_updated":"2025-04-25T08:06:22Z","snapshot_observed_at":"2026-08-18T08:38:54.349344Z","submitted_at":"2025-04-25T08:06:22Z","title":"EcoServe: Enabling Cost-effective LLM Serving with Proactive Intra- and Inter-Instance Orchestration","version":1},"cited_work":{"arxiv_id":"2504.18154","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2504.18154","snapshot_observed_at":"2026-07-02T15:27:06.035370Z","title":"arXiv preprint arXiv:2504.18154 , year=","venue":null,"work_id":"9194ca84-f774-433b-8d5f-7cd9f1074f2f","year":2025},"citing_paper":{"arxiv_id":"2606.05933","last_updated":"2026-06-04T09:36:40Z","snapshot_observed_at":"2026-08-13T14:18:06.635132Z","submitted_at":"2026-06-04T09:36:40Z","title":"Beyond Greedy Chunking: SLO-Aware Sliding-Window Scheduling for LLM Inference","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-06-27T23:49:28.318260Z"},"links":{"cited_paper":"/paper/2504.18154","citing_paper":"/paper/2606.05933"},"observation_digest":"sha256:4e88e48281eb32cae21cc2bcba1bf06330ecbd936ac20054d668108fd8efd1c0","observation_id":"d5da95d9-fc98-450b-8e17-40713bbb4b43","resolution":{"observed_at":"2026-07-02T15:27:06.037068Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2504.18154/citation-record","integrity":"/paper/2504.18154/integrity","json":"/paper/2504.18154/citation-record.json","paper":"/paper/2504.18154"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:28:53.070503Z","title":"Github Copilot","venue":null,"work_id":"90e69bad-3c39-431d-9383-6c5459e3f9a0","year":2023},"citing_paper":{"arxiv_id":"2504.18154","last_updated":"2025-04-25T08:06:22Z","snapshot_observed_at":"2026-08-18T08:38:54.349344Z","submitted_at":"2025-04-25T08:06:22Z","title":"EcoServe: Enabling Cost-effective LLM Serving with Proactive Intra- and Inter-Instance Orchestration","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-16T10:28:52.343695Z"},"links":{"citing_paper":"/paper/2504.18154"},"observation_digest":"sha256:603a514aaeb11f34fe038106ccf080270e430be4347d8419710fff5bf3c6d316","observation_id":"90a39ebd-a583-4439-af30-167a617cd87d","resolution":{"observed_at":"2026-08-16T10:28:53.074202Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:28:53.057456Z","title":null,"venue":null,"work_id":"617b357b-ef14-48f9-a0c3-0cc4171e9953","year":2024},"citing_paper":{"arxiv_id":"2504.18154","last_updated":"2025-04-25T08:06:22Z","snapshot_observed_at":"2026-08-18T08:38:54.349344Z","submitted_at":"2025-04-25T08:06:22Z","title":"EcoServe: Enabling Cost-effective LLM Serving with Proactive Intra- and Inter-Instance Orchestration","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-16T10:28:52.348035Z"},"links":{"citing_paper":"/paper/2504.18154"},"observation_digest":"sha256:c91441c1037bddd72e2973e7acf857c35a87e8de81df0498dee6c1643b48f25d","observation_id":"5059a1f8-064c-42f6-a978-529a02ca2679","resolution":{"observed_at":"2026-08-16T10:28:53.062322Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:28:53.046889Z","title":"Faster Transformer","venue":null,"work_id":"d66b6111-c479-4998-8d0e-f975839197f7","year":2024},"citing_paper":{"arxiv_id":"2504.18154","last_updated":"2025-04-25T08:06:22Z","snapshot_observed_at":"2026-08-18T08:38:54.349344Z","submitted_at":"2025-04-25T08:06:22Z","title":"EcoServe: Enabling Cost-effective LLM Serving with Proactive Intra- and Inter-Instance Orchestration","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-16T10:28:52.351698Z"},"links":{"citing_paper":"/paper/2504.18154"},"observation_digest":"sha256:30b0dc0fe3b9e1e8a14d1808eab362c540aa2a87bb2bd42d43d4c3c51297c5c4","observation_id":"fab712f0-b525-4fc8-bf53-3cb1b8369fd3","resolution":{"observed_at":"2026-08-16T10:28:53.050116Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:28:53.034612Z","title":null,"venue":null,"work_id":"058d9a09-21d7-4ff6-aaee-90f7f683b2a7","year":2024},"citing_paper":{"arxiv_id":"2504.18154","last_updated":"2025-04-25T08:06:22Z","snapshot_observed_at":"2026-08-18T08:38:54.349344Z","submitted_at":"2025-04-25T08:06:22Z","title":"EcoServe: Enabling Cost-effective LLM Serving with Proactive Intra- and Inter-Instance Orchestration","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-16T10:28:52.357270Z"},"links":{"citing_paper":"/paper/2504.18154"},"observation_digest":"sha256:86782e2ddd84eb094c93d0a8651ff7522a1c78809232872c361eb1f3c28186d3","observation_id":"1a512663-780a-4052-830e-acff74e3c9a0","resolution":{"observed_at":"2026-08-16T10:28:53.038963Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:28:53.021906Z","title":"vllm: Easy, fast, and cheap llm serving for everyone","venue":null,"work_id":"0ebc011f-138f-44f1-87db-fe93bc3f1329","year":2024},"citing_paper":{"arxiv_id":"2504.18154","last_updated":"2025-04-25T08:06:22Z","snapshot_observed_at":"2026-08-18T08:38:54.349344Z","submitted_at":"2025-04-25T08:06:22Z","title":"EcoServe: Enabling Cost-effective LLM Serving with Proactive Intra- and Inter-Instance Orchestration","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-16T10:28:52.361285Z"},"links":{"citing_paper":"/paper/2504.18154"},"observation_digest":"sha256:e56e6aab900ee9a73e6b1f1d93850670a026f01ef9210dc0d0fa81c18e8b854d","observation_id":"ab5b784d-8537-453d-bd18-fe746965d4ca","resolution":{"observed_at":"2026-08-16T10:28:53.026748Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:28:53.008854Z","title":"Character ai","venue":null,"work_id":"c0f00bc3-e8eb-4c4b-9878-994df981f591","year":2025},"citing_paper":{"arxiv_id":"2504.18154","last_updated":"2025-04-25T08:06:22Z","snapshot_observed_at":"2026-08-18T08:38:54.349344Z","submitted_at":"2025-04-25T08:06:22Z","title":"EcoServe: Enabling Cost-effective LLM Serving with Proactive Intra- and Inter-Instance Orchestration","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-16T10:28:52.364776Z"},"links":{"citing_paper":"/paper/2504.18154"},"observation_digest":"sha256:881bf92114aa1303f5970d569d086dff1c7d1f33bee8d96c160cfe7aa56ba633","observation_id":"790118fd-8436-4c32-b49c-22a6419ab4cc","resolution":{"observed_at":"2026-08-16T10:28:53.012671Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:28:52.997602Z","title":null,"venue":null,"work_id":"23da2bf4-5cf5-458e-a2a6-36fbe7512b7f","year":2025},"citing_paper":{"arxiv_id":"2504.18154","last_updated":"2025-04-25T08:06:22Z","snapshot_observed_at":"2026-08-18T08:38:54.349344Z","submitted_at":"2025-04-25T08:06:22Z","title":"EcoServe: Enabling Cost-effective LLM Serving with Proactive Intra- and Inter-Instance Orchestration","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-16T10:28:52.369294Z"},"links":{"citing_paper":"/paper/2504.18154"},"observation_digest":"sha256:0fc8f1a130fa79f5ae4a50df9bb307e771d29e5e5fbd37aec16633b18015ff5a","observation_id":"b45f90ec-263e-4b9c-bfb9-e73fa2baf704","resolution":{"observed_at":"2026-08-16T10:28:53.001134Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:28:52.987091Z","title":null,"venue":null,"work_id":"bb1c3c8e-de3b-4f6d-a715-8fa1ff1ee66b","year":2024},"citing_paper":{"arxiv_id":"2504.18154","last_updated":"2025-04-25T08:06:22Z","snapshot_observed_at":"2026-08-18T08:38:54.349344Z","submitted_at":"2025-04-25T08:06:22Z","title":"EcoServe: Enabling Cost-effective LLM Serving with Proactive Intra- and Inter-Instance Orchestration","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-16T10:28:52.372385Z"},"links":{"citing_paper":"/paper/2504.18154"},"observation_digest":"sha256:0229b1b4ede872c8c4d31d142e350cba20c782c7b585f749faad292edcc669f6","observation_id":"63ba6196-67ee-46b0-a317-ee3dbf7bc07d","resolution":{"observed_at":"2026-08-16T10:28:52.990574Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:28:52.974871Z","title":null,"venue":null,"work_id":"5a041ce4-1dfb-4e81-a8e3-d57f949fcef1","year":2024},"citing_paper":{"arxiv_id":"2504.18154","last_updated":"2025-04-25T08:06:22Z","snapshot_observed_at":"2026-08-18T08:38:54.349344Z","submitted_at":"2025-04-25T08:06:22Z","title":"EcoServe: Enabling Cost-effective LLM Serving with Proactive Intra- and Inter-Instance Orchestration","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-16T10:28:52.376638Z"},"links":{"citing_paper":"/paper/2504.18154"},"observation_digest":"sha256:34d531276594733ed7bf11acd1fe0581fb072d71b70340b9d5b49d87cf247fcc","observation_id":"31b37b73-48f4-40b5-bf81-a3b19efa2577","resolution":{"observed_at":"2026-08-16T10:28:52.979332Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.13245","last_updated":"2023-12-23T17:55:11Z","snapshot_observed_at":"2026-08-15T06:28:09.529747Z","submitted_at":"2023-05-22T17:16:38Z","title":"GQA: Training Generalized Multi-Query Transformer Models from Multi-Head Checkpoints","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.13245","snapshot_observed_at":"2026-08-16T10:28:52.380722Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2504.18154","last_updated":"2025-04-25T08:06:22Z","snapshot_observed_at":"2026-08-18T08:38:54.349344Z","submitted_at":"2025-04-25T08:06:22Z","title":"EcoServe: Enabling Cost-effective LLM Serving with Proactive Intra- and Inter-Instance Orchestration","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-16T10:28:52.380722Z"},"links":{"cited_paper":"/paper/2305.13245","citing_paper":"/paper/2504.18154"},"observation_digest":"sha256:0d8b6d7d390078e1ccd66d9c6833fe65e786f1b32ce10250f940209ee20684d5","observation_id":"faaf55e2-8134-4571-9111-82eb45256317","resolution":{"observed_at":"2026-08-16T10:28:52.380722Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:28:52.965670Z","title":null,"venue":null,"work_id":"bd1c02c1-7882-4e69-ba0e-bb49df3b8b57","year":2020},"citing_paper":{"arxiv_id":"2504.18154","last_updated":"2025-04-25T08:06:22Z","snapshot_observed_at":"2026-08-18T08:38:54.349344Z","submitted_at":"2025-04-25T08:06:22Z","title":"EcoServe: Enabling Cost-effective LLM Serving with Proactive Intra- and Inter-Instance Orchestration","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-16T10:28:52.384850Z"},"links":{"citing_paper":"/paper/2504.18154"},"observation_digest":"sha256:7ca91fae0a442a0c8eeba4bdeaa06cd64212ed980a5d79ccc7336382428b744b","observation_id":"b595d392-de6c-42da-b95e-379b17254809","resolution":{"observed_at":"2026-08-16T10:28:52.968749Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:28:52.388729Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2504.18154","last_updated":"2025-04-25T08:06:22Z","snapshot_observed_at":"2026-08-18T08:38:54.349344Z","submitted_at":"2025-04-25T08:06:22Z","title":"EcoServe: Enabling Cost-effective LLM Serving with Proactive Intra- and Inter-Instance Orchestration","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-16T10:28:52.388729Z"},"links":{"citing_paper":"/paper/2504.18154"},"observation_digest":"sha256:e7eff7ce977667287e9bd003eeefb79e90b44e6a603284bbf8871e3831f5df64","observation_id":"c6b5d3c4-32d2-46f3-a15e-c7a664abaa99","resolution":{"observed_at":"2026-08-16T10:28:52.388729Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2301.08658","last_updated":"2023-01-20T16:06:59Z","snapshot_observed_at":"2026-08-16T16:01:08.066346Z","submitted_at":"2023-01-20T16:06:59Z","title":"ATP: Adaptive Tensor Parallelism for Foundation Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2301.08658","snapshot_observed_at":"2026-08-16T10:28:52.392422Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2504.18154","last_updated":"2025-04-25T08:06:22Z","snapshot_observed_at":"2026-08-18T08:38:54.349344Z","submitted_at":"2025-04-25T08:06:22Z","title":"EcoServe: Enabling Cost-effective LLM Serving with Proactive Intra- and Inter-Instance Orchestration","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-16T10:28:52.392422Z"},"links":{"cited_paper":"/paper/2301.08658","citing_paper":"/paper/2504.18154"},"observation_digest":"sha256:41785fdbb371119a4bee1793916901afa2dd34de6115efe43b1d19e9de7c9796","observation_id":"1591e0fc-6ccd-4224-b9d7-2d3f8e129e9c","resolution":{"observed_at":"2026-08-16T10:28:52.392422Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:28:52.397103Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2504.18154","last_updated":"2025-04-25T08:06:22Z","snapshot_observed_at":"2026-08-18T08:38:54.349344Z","submitted_at":"2025-04-25T08:06:22Z","title":"EcoServe: Enabling Cost-effective LLM Serving with Proactive Intra- and Inter-Instance Orchestration","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-16T10:28:52.397103Z"},"links":{"citing_paper":"/paper/2504.18154"},"observation_digest":"sha256:62d464b1487e72dfb838f9551e32bc28ea4ff3de474948db851818a0b3282a46","observation_id":"52a894be-3331-4b5d-b7d1-eb7742c4810f","resolution":{"observed_at":"2026-08-16T10:28:52.397103Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:28:52.937169Z","title":null,"venue":null,"work_id":"1bedf596-37cd-48c2-8953-b9e6fad83f40","year":2024},"citing_paper":{"arxiv_id":"2504.18154","last_updated":"2025-04-25T08:06:22Z","snapshot_observed_at":"2026-08-18T08:38:54.349344Z","submitted_at":"2025-04-25T08:06:22Z","title":"EcoServe: Enabling Cost-effective LLM Serving with Proactive Intra- and Inter-Instance Orchestration","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-16T10:28:52.403130Z"},"links":{"citing_paper":"/paper/2504.18154"},"observation_digest":"sha256:7b422eaf8b1eab7b0d74349024771c248c0373c70ea39ea8c2e6911bf8936667","observation_id":"4aa76aa4-d08c-46e8-b544-cbac8bbbd9a8","resolution":{"observed_at":"2026-08-16T10:28:52.940381Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:28:52.926860Z","title":null,"venue":null,"work_id":"a0e8adbc-89df-497a-b69a-9cf44d90c82e","year":2024},"citing_paper":{"arxiv_id":"2504.18154","last_updated":"2025-04-25T08:06:22Z","snapshot_observed_at":"2026-08-18T08:38:54.349344Z","submitted_at":"2025-04-25T08:06:22Z","title":"EcoServe: Enabling Cost-effective LLM Serving with Proactive Intra- and Inter-Instance Orchestration","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-16T10:28:52.406750Z"},"links":{"citing_paper":"/paper/2504.18154"},"observation_digest":"sha256:9e06037d855726b6d666cef2c2ce00da466604c4b3dea6ac5231c3695e9e371f","observation_id":"4eae3516-1b02-432f-9d2d-8d83cea8777d","resolution":{"observed_at":"2026-08-16T10:28:52.929393Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:28:52.913758Z","title":null,"venue":null,"work_id":"6a6cfddb-a637-4da0-bec3-85a50df2c783","year":2024},"citing_paper":{"arxiv_id":"2504.18154","last_updated":"2025-04-25T08:06:22Z","snapshot_observed_at":"2026-08-18T08:38:54.349344Z","submitted_at":"2025-04-25T08:06:22Z","title":"EcoServe: Enabling Cost-effective LLM Serving with Proactive Intra- and Inter-Instance Orchestration","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-16T10:28:52.409775Z"},"links":{"citing_paper":"/paper/2504.18154"},"observation_digest":"sha256:fe4c1eb008ded7ec211434adfcf30d5047c095df7dc370d184772b81807baa13","observation_id":"2f0a4305-3a41-4a37-b606-644bc8599afb","resolution":{"observed_at":"2026-08-16T10:28:52.918547Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.21783","last_updated":"2024-11-23T23:27:33Z","snapshot_observed_at":"2026-08-13T17:20:44.002518Z","submitted_at":"2024-07-31T17:54:27Z","title":"The Llama 3 Herd of Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.21783","snapshot_observed_at":"2026-08-16T10:28:52.412545Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2504.18154","last_updated":"2025-04-25T08:06:22Z","snapshot_observed_at":"2026-08-18T08:38:54.349344Z","submitted_at":"2025-04-25T08:06:22Z","title":"EcoServe: Enabling Cost-effective LLM Serving with Proactive Intra- and Inter-Instance Orchestration","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-16T10:28:52.412545Z"},"links":{"cited_paper":"/paper/2407.21783","citing_paper":"/paper/2504.18154"},"observation_digest":"sha256:9ceb98f84ed32d9bda1cc24d11c5723c51f78a05d8ba15e2d219f610c8dda50e","observation_id":"6e8ce1a6-223e-453a-9be0-50234185dc2a","resolution":{"observed_at":"2026-08-16T10:28:52.412545Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.11421","last_updated":"2024-03-18T02:30:23Z","snapshot_observed_at":"2026-08-16T14:08:58.432540Z","submitted_at":"2024-03-18T02:30:23Z","title":"FastDecode: High-Throughput GPU-Efficient LLM Serving using Heterogeneous Pipelines","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.11421","snapshot_observed_at":"2026-08-16T10:28:52.416961Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2504.18154","last_updated":"2025-04-25T08:06:22Z","snapshot_observed_at":"2026-08-18T08:38:54.349344Z","submitted_at":"2025-04-25T08:06:22Z","title":"EcoServe: Enabling Cost-effective LLM Serving with Proactive Intra- and Inter-Instance Orchestration","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-16T10:28:52.416961Z"},"links":{"cited_paper":"/paper/2403.11421","citing_paper":"/paper/2504.18154"},"observation_digest":"sha256:84f8a2953e7e0d3f14718c3ef5030fc0efd7cc0e402f87b0529375bfaf6dfc2c","observation_id":"cfd21116-0766-43e2-8033-d0362687c291","resolution":{"observed_at":"2026-08-16T10:28:52.416961Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:28:52.903834Z","title":null,"venue":null,"work_id":"ffdf1f15-f396-4557-943f-b61c052f1096","year":2019},"citing_paper":{"arxiv_id":"2504.18154","last_updated":"2025-04-25T08:06:22Z","snapshot_observed_at":"2026-08-18T08:38:54.349344Z","submitted_at":"2025-04-25T08:06:22Z","title":"EcoServe: Enabling Cost-effective LLM Serving with Proactive Intra- and Inter-Instance Orchestration","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-16T10:28:52.421192Z"},"links":{"citing_paper":"/paper/2504.18154"},"observation_digest":"sha256:8b3f278a0aa2b2e2982afaa38a53623e3729ce481a5939cb03c7e263b20a0b0b","observation_id":"21829075-7633-4d3c-af9d-45af183fbc78","resolution":{"observed_at":"2026-08-16T10:28:52.907011Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:28:52.424515Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2504.18154","last_updated":"2025-04-25T08:06:22Z","snapshot_observed_at":"2026-08-18T08:38:54.349344Z","submitted_at":"2025-04-25T08:06:22Z","title":"EcoServe: Enabling Cost-effective LLM Serving with Proactive Intra- and Inter-Instance Orchestration","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-16T10:28:52.424515Z"},"links":{"citing_paper":"/paper/2504.18154"},"observation_digest":"sha256:90f75f82bebaa64272e424179cad290d66563e4ef32742fe641b89621f807530","observation_id":"23fc55cc-75ad-4266-b0ae-58b5aae2ec77","resolution":{"observed_at":"2026-08-16T10:28:52.424515Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2309.14509","last_updated":"2023-10-04T16:51:13Z","snapshot_observed_at":"2026-08-13T22:49:56.824755Z","submitted_at":"2023-09-25T20:15:57Z","title":"DeepSpeed Ulysses: System Optimizations for Enabling Training of Extreme Long Sequence Transformer Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.14509","snapshot_observed_at":"2026-08-16T10:28:52.428447Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2504.18154","last_updated":"2025-04-25T08:06:22Z","snapshot_observed_at":"2026-08-18T08:38:54.349344Z","submitted_at":"2025-04-25T08:06:22Z","title":"EcoServe: Enabling Cost-effective LLM Serving with Proactive Intra- and Inter-Instance Orchestration","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-16T10:28:52.428447Z"},"links":{"cited_paper":"/paper/2309.14509","citing_paper":"/paper/2504.18154"},"observation_digest":"sha256:cc3c53ee5d1b90763a0ef11f65ddc4aef870237320aacc70ec1ec54099d988a6","observation_id":"8b828604-8365-4bd5-be17-a634221b8da6","resolution":{"observed_at":"2026-08-16T10:28:52.428447Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.12457","last_updated":"2024-04-25T06:47:57Z","snapshot_observed_at":"2026-08-18T08:35:06.332399Z","submitted_at":"2024-04-18T18:32:30Z","title":"RAGCache: Efficient Knowledge Caching for Retrieval-Augmented Generation","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.12457","snapshot_observed_at":"2026-08-16T10:28:52.431819Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2504.18154","last_updated":"2025-04-25T08:06:22Z","snapshot_observed_at":"2026-08-18T08:38:54.349344Z","submitted_at":"2025-04-25T08:06:22Z","title":"EcoServe: Enabling Cost-effective LLM Serving with Proactive Intra- and Inter-Instance Orchestration","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-16T10:28:52.431819Z"},"links":{"cited_paper":"/paper/2404.12457","citing_paper":"/paper/2504.18154"},"observation_digest":"sha256:4357a97331d8cb4e91b08b40f092d8cfb3ca293c38b1654de2f654cb798cf9a1","observation_id":"ace5662a-dee5-4e51-b39a-3b68e60323eb","resolution":{"observed_at":"2026-08-16T10:28:52.431819Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:28:52.435320Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2504.18154","last_updated":"2025-04-25T08:06:22Z","snapshot_observed_at":"2026-08-18T08:38:54.349344Z","submitted_at":"2025-04-25T08:06:22Z","title":"EcoServe: Enabling Cost-effective LLM Serving with Proactive Intra- and Inter-Instance Orchestration","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-16T10:28:52.435320Z"},"links":{"citing_paper":"/paper/2504.18154"},"observation_digest":"sha256:d662889e065a288da402dca3a1d1c89d0baa9a7e4207a7f3a46647b2d6aafc64","observation_id":"8c113073-17c9-481e-8dce-b80dbd375ffe","resolution":{"observed_at":"2026-08-16T10:28:52.435320Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:28:52.874441Z","title":null,"venue":null,"work_id":"69387db0-cfb8-4ba4-b2da-f9e23740f4fb","year":2023},"citing_paper":{"arxiv_id":"2504.18154","last_updated":"2025-04-25T08:06:22Z","snapshot_observed_at":"2026-08-18T08:38:54.349344Z","submitted_at":"2025-04-25T08:06:22Z","title":"EcoServe: Enabling Cost-effective LLM Serving with Proactive Intra- and Inter-Instance Orchestration","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-16T10:28:52.442498Z"},"links":{"citing_paper":"/paper/2504.18154"},"observation_digest":"sha256:38c76ed77b26ecb5a6838c267f59fea893dda8d5467fedab98f9d391aa257a6c","observation_id":"f5d524ad-90ae-4235-a3d3-843ab4fcd47e","resolution":{"observed_at":"2026-08-16T10:28:52.878056Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:28:52.864495Z","title":null,"venue":null,"work_id":"ba5c095e-5596-4c11-87ca-c2b1c18067c0","year":2021},"citing_paper":{"arxiv_id":"2504.18154","last_updated":"2025-04-25T08:06:22Z","snapshot_observed_at":"2026-08-18T08:38:54.349344Z","submitted_at":"2025-04-25T08:06:22Z","title":"EcoServe: Enabling Cost-effective LLM Serving with Proactive Intra- and Inter-Instance Orchestration","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-16T10:28:52.445385Z"},"links":{"citing_paper":"/paper/2504.18154"},"observation_digest":"sha256:2cda6b9b6f9ea4133a9e4b8dabbd4efbe27348c895af42c2e8611580197db812","observation_id":"de6751d6-02c2-4e49-bcbb-c3529214b869","resolution":{"observed_at":"2026-08-16T10:28:52.867240Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.20552","last_updated":"2025-03-26T13:48:35Z","snapshot_observed_at":"2026-08-18T09:23:54.569858Z","submitted_at":"2025-03-26T13:48:35Z","title":"Injecting Adrenaline into LLM Serving: Boosting Resource Utilization and Throughput via Attention Disaggregation","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.20552","snapshot_observed_at":"2026-08-16T10:28:52.448573Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2504.18154","last_updated":"2025-04-25T08:06:22Z","snapshot_observed_at":"2026-08-18T08:38:54.349344Z","submitted_at":"2025-04-25T08:06:22Z","title":"EcoServe: Enabling Cost-effective LLM Serving with Proactive Intra- and Inter-Instance Orchestration","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-16T10:28:52.448573Z"},"links":{"cited_paper":"/paper/2503.20552","citing_paper":"/paper/2504.18154"},"observation_digest":"sha256:655718dc1624702173f69b2180764342882add76f5c06e753ac885f3ec39db36","observation_id":"52c91951-a885-4c51-9254-d96119a2c3fc","resolution":{"observed_at":"2026-08-16T10:28:52.448573Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.02669","last_updated":"2024-07-04T15:12:54Z","snapshot_observed_at":"2026-08-17T04:58:55.259763Z","submitted_at":"2024-01-05T06:53:00Z","title":"Infinite-LLM: Efficient LLM Service for Long Context with DistAttention and Distributed KVCache","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.02669","snapshot_observed_at":"2026-08-16T10:28:52.452135Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2504.18154","last_updated":"2025-04-25T08:06:22Z","snapshot_observed_at":"2026-08-18T08:38:54.349344Z","submitted_at":"2025-04-25T08:06:22Z","title":"EcoServe: Enabling Cost-effective LLM Serving with Proactive Intra- and Inter-Instance Orchestration","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-16T10:28:52.452135Z"},"links":{"cited_paper":"/paper/2401.02669","citing_paper":"/paper/2504.18154"},"observation_digest":"sha256:51695c8a3065794898a5aeec01c863d501ec0b18aeacac1e510c3ab0123dfa61","observation_id":"0c40a79b-c316-4493-b732-015a2ccfaa86","resolution":{"observed_at":"2026-08-16T10:28:52.452135Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.01889","last_updated":"2023-11-27T06:38:47Z","snapshot_observed_at":"2026-08-14T10:14:18.862721Z","submitted_at":"2023-10-03T08:44:50Z","title":"Ring Attention with Blockwise Transformers for Near-Infinite Context","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.01889","snapshot_observed_at":"2026-08-16T10:28:52.455554Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2504.18154","last_updated":"2025-04-25T08:06:22Z","snapshot_observed_at":"2026-08-18T08:38:54.349344Z","submitted_at":"2025-04-25T08:06:22Z","title":"EcoServe: Enabling Cost-effective LLM Serving with Proactive Intra- and Inter-Instance Orchestration","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-16T10:28:52.455554Z"},"links":{"cited_paper":"/paper/2310.01889","citing_paper":"/paper/2504.18154"},"observation_digest":"sha256:8480b5ac31c267edd6ff54c158fd165bf90f45999ab1f883c2136963fa1ab212","observation_id":"c7480dc0-b0be-46f6-bf97-984c48aa59d4","resolution":{"observed_at":"2026-08-16T10:28:52.455554Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.05821","last_updated":"2025-04-09T10:23:39Z","snapshot_observed_at":"2026-08-16T14:11:24.750510Z","submitted_at":"2024-03-09T07:01:44Z","title":"Optimizing LLM Queries in Relational Data Analytics Workloads","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.05821","snapshot_observed_at":"2026-08-16T10:28:52.460076Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2504.18154","last_updated":"2025-04-25T08:06:22Z","snapshot_observed_at":"2026-08-18T08:38:54.349344Z","submitted_at":"2025-04-25T08:06:22Z","title":"EcoServe: Enabling Cost-effective LLM Serving with Proactive Intra- and Inter-Instance Orchestration","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-16T10:28:52.460076Z"},"links":{"cited_paper":"/paper/2403.05821","citing_paper":"/paper/2504.18154"},"observation_digest":"sha256:1eab0e0f979381bf447dd544cd83646b3c40a3e432270f7a6f7b68c8684db84b","observation_id":"eb1afff9-f975-4e09-bc0f-d9cac7d7e3a9","resolution":{"observed_at":"2026-08-16T10:28:52.460076Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.09781","last_updated":"2024-04-01T02:18:42Z","snapshot_observed_at":"2026-08-16T15:32:32.643405Z","submitted_at":"2023-05-16T20:12:59Z","title":"SpecInfer: Accelerating Generative Large Language Model Serving with Tree-based Speculative Inference and Verification","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.09781","snapshot_observed_at":"2026-08-16T10:28:52.463912Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2504.18154","last_updated":"2025-04-25T08:06:22Z","snapshot_observed_at":"2026-08-18T08:38:54.349344Z","submitted_at":"2025-04-25T08:06:22Z","title":"EcoServe: Enabling Cost-effective LLM Serving with Proactive Intra- and Inter-Instance Orchestration","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-16T10:28:52.463912Z"},"links":{"cited_paper":"/paper/2305.09781","citing_paper":"/paper/2504.18154"},"observation_digest":"sha256:48b47f4feaf2709db50f0420807859740d042b82f9525e1b32ecf2548dcbd77a","observation_id":"073825de-aa8c-4835-8b5b-32cfa0e50365","resolution":{"observed_at":"2026-08-16T10:28:52.463912Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:28:52.468739Z","title":null,"venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2504.18154","last_updated":"2025-04-25T08:06:22Z","snapshot_observed_at":"2026-08-18T08:38:54.349344Z","submitted_at":"2025-04-25T08:06:22Z","title":"EcoServe: Enabling Cost-effective LLM Serving with Proactive Intra- and Inter-Instance Orchestration","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-16T10:28:52.468739Z"},"links":{"citing_paper":"/paper/2504.18154"},"observation_digest":"sha256:f04679e3fd381c4db0bdcaf94acf502762ee7c922c8aedf563ee8bbee8fc79ab","observation_id":"61af3c15-2778-43f5-8733-8aca9d9d2dd0","resolution":{"observed_at":"2026-08-16T10:28:52.468739Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:28:52.471742Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2504.18154","last_updated":"2025-04-25T08:06:22Z","snapshot_observed_at":"2026-08-18T08:38:54.349344Z","submitted_at":"2025-04-25T08:06:22Z","title":"EcoServe: Enabling Cost-effective LLM Serving with Proactive Intra- and Inter-Instance Orchestration","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-16T10:28:52.471742Z"},"links":{"citing_paper":"/paper/2504.18154"},"observation_digest":"sha256:0d22e22caa747f373835901b285c9296352d7c39e3984b60c7858addc26dbce8","observation_id":"28e44035-5ae6-4119-ad1c-70d4a5d5ae17","resolution":{"observed_at":"2026-08-16T10:28:52.471742Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.04437","last_updated":"2025-01-29T04:10:41Z","snapshot_observed_at":"2026-08-16T13:54:36.547025Z","submitted_at":"2024-05-07T16:00:32Z","title":"vAttention: Dynamic Memory Management for Serving LLMs without PagedAttention","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.04437","snapshot_observed_at":"2026-08-16T10:28:52.475171Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2504.18154","last_updated":"2025-04-25T08:06:22Z","snapshot_observed_at":"2026-08-18T08:38:54.349344Z","submitted_at":"2025-04-25T08:06:22Z","title":"EcoServe: Enabling Cost-effective LLM Serving with Proactive Intra- and Inter-Instance Orchestration","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-16T10:28:52.475171Z"},"links":{"cited_paper":"/paper/2405.04437","citing_paper":"/paper/2504.18154"},"observation_digest":"sha256:774a8a190edc6d6049b552817333bf4657a02ca36b8045e4a07dddb8d0269c28","observation_id":"a537719b-59a0-4ae9-bc80-d2e88cd0118b","resolution":{"observed_at":"2026-08-16T10:28:52.475171Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:28:52.843374Z","title":null,"venue":null,"work_id":"80b664af-cd01-4d50-ae31-e4b45ede3d4d","year":2025},"citing_paper":{"arxiv_id":"2504.18154","last_updated":"2025-04-25T08:06:22Z","snapshot_observed_at":"2026-08-18T08:38:54.349344Z","submitted_at":"2025-04-25T08:06:22Z","title":"EcoServe: Enabling Cost-effective LLM Serving with Proactive Intra- and Inter-Instance Orchestration","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-16T10:28:52.479248Z"},"links":{"citing_paper":"/paper/2504.18154"},"observation_digest":"sha256:ddd3b14aa136daba494194e3819733396cf8c8308c32c437ab136dfabbd6a069","observation_id":"f39fbd49-d909-4253-8707-22dfcb49cbe4","resolution":{"observed_at":"2026-08-16T10:28:52.847119Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.12950","last_updated":"2024-01-31T19:47:26Z","snapshot_observed_at":"2026-08-18T03:59:39.242039Z","submitted_at":"2023-08-24T17:39:13Z","title":"Code Llama: Open Foundation Models for Code","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.12950","snapshot_observed_at":"2026-08-16T10:28:52.482825Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2504.18154","last_updated":"2025-04-25T08:06:22Z","snapshot_observed_at":"2026-08-18T08:38:54.349344Z","submitted_at":"2025-04-25T08:06:22Z","title":"EcoServe: Enabling Cost-effective LLM Serving with Proactive Intra- and Inter-Instance Orchestration","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-16T10:28:52.482825Z"},"links":{"cited_paper":"/paper/2308.12950","citing_paper":"/paper/2504.18154"},"observation_digest":"sha256:9a082443d986d498d7e192af3a1933f644e84afa0d8a81b13d5df3c2f0b53062","observation_id":"c2934c1c-bea9-4202-96db-0cd6f364d644","resolution":{"observed_at":"2026-08-16T10:28:52.482825Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:28:52.830530Z","title":null,"venue":null,"work_id":"8ae32bd8-ac74-43b5-8e9e-ddaae5af28f9","year":1911},"citing_paper":{"arxiv_id":"2504.18154","last_updated":"2025-04-25T08:06:22Z","snapshot_observed_at":"2026-08-18T08:38:54.349344Z","submitted_at":"2025-04-25T08:06:22Z","title":"EcoServe: Enabling Cost-effective LLM Serving with Proactive Intra- and Inter-Instance Orchestration","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-16T10:28:52.487089Z"},"links":{"citing_paper":"/paper/2504.18154"},"observation_digest":"sha256:95a77cf59240aaa620259818cc9a658f37cc5d651171c1d73f4bf01f4d9b6d1f","observation_id":"8633ef53-d4b1-41e2-87df-eb6b3d4b45fe","resolution":{"observed_at":"2026-08-16T10:28:52.834033Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:28:52.819118Z","title":null,"venue":null,"work_id":"82326148-c0d0-4d4b-a630-f73aedb3717d","year":2018},"citing_paper":{"arxiv_id":"2504.18154","last_updated":"2025-04-25T08:06:22Z","snapshot_observed_at":"2026-08-18T08:38:54.349344Z","submitted_at":"2025-04-25T08:06:22Z","title":"EcoServe: Enabling Cost-effective LLM Serving with Proactive Intra- and Inter-Instance Orchestration","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-16T10:28:52.490773Z"},"links":{"citing_paper":"/paper/2504.18154"},"observation_digest":"sha256:306a361214e8ae64acf2e8a76bb08155f8ded7acfd2291306549f58f87f7dee0","observation_id":"fb1fe424-1ae6-4692-b090-bd759cb7a7c8","resolution":{"observed_at":"2026-08-16T10:28:52.822623Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:28:52.493720Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2504.18154","last_updated":"2025-04-25T08:06:22Z","snapshot_observed_at":"2026-08-18T08:38:54.349344Z","submitted_at":"2025-04-25T08:06:22Z","title":"EcoServe: Enabling Cost-effective LLM Serving with Proactive Intra- and Inter-Instance Orchestration","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-16T10:28:52.493720Z"},"links":{"citing_paper":"/paper/2504.18154"},"observation_digest":"sha256:5e197a7106a879d4277dfd7e2df20915d7b1cba374be1f8c1eaca4fcd5309b09","observation_id":"a2ef98f4-9840-4f28-b0da-176a4f042c2c","resolution":{"observed_at":"2026-08-16T10:28:52.493720Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1909.08053","last_updated":"2020-03-13T23:45:18Z","snapshot_observed_at":"2026-08-12T10:50:46.357243Z","submitted_at":"2019-09-17T19:42:54Z","title":"Megatron-LM: Training Multi-Billion Parameter Language Models Using Model Parallelism","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1909.08053","snapshot_observed_at":"2026-08-16T10:28:52.501489Z","title":null,"venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2504.18154","last_updated":"2025-04-25T08:06:22Z","snapshot_observed_at":"2026-08-18T08:38:54.349344Z","submitted_at":"2025-04-25T08:06:22Z","title":"EcoServe: Enabling Cost-effective LLM Serving with Proactive Intra- and Inter-Instance Orchestration","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-16T10:28:52.501489Z"},"links":{"cited_paper":"/paper/1909.08053","citing_paper":"/paper/2504.18154"},"observation_digest":"sha256:b6b14fbb72e568735d7ea4aa46dd9d90e597792eac0fa0466295dc3209ac8f17","observation_id":"55d5ab65-cb75-474f-abd1-6b182eeba055","resolution":{"observed_at":"2026-08-16T10:28:52.501489Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:28:52.796673Z","title":null,"venue":null,"work_id":"b4bd2c71-96ae-418f-813a-926abc8a5fdb","year":2023},"citing_paper":{"arxiv_id":"2504.18154","last_updated":"2025-04-25T08:06:22Z","snapshot_observed_at":"2026-08-18T08:38:54.349344Z","submitted_at":"2025-04-25T08:06:22Z","title":"EcoServe: Enabling Cost-effective LLM Serving with Proactive Intra- and Inter-Instance Orchestration","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-16T10:28:52.505275Z"},"links":{"citing_paper":"/paper/2504.18154"},"observation_digest":"sha256:1a6b67b30c92472b4795e2f93c720ae19f24eacf7f800ab173402637bf8f5208","observation_id":"45f88ba3-0569-4aa3-bf0f-d957b6b02bd4","resolution":{"observed_at":"2026-08-16T10:28:52.799812Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:28:52.498170Z","title":"In International Conference on Machine Learning","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2504.18154","last_updated":"2025-04-25T08:06:22Z","snapshot_observed_at":"2026-08-18T08:38:54.349344Z","submitted_at":"2025-04-25T08:06:22Z","title":"EcoServe: Enabling Cost-effective LLM Serving with Proactive Intra- and Inter-Instance Orchestration","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-16T10:28:52.498170Z"},"links":{"citing_paper":"/paper/2504.18154"},"observation_digest":"sha256:e030febe2866ac65d24f2d552e9cfd4adcbb732c1def12d357ce7cc231a6e086","observation_id":"a24904de-6523-46b2-872a-a2dcb7bdf45e","resolution":{"observed_at":"2026-08-16T10:28:52.498170Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2302.13971","last_updated":"2023-02-27T17:11:15Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-02-27T17:11:15Z","title":"LLaMA: Open and Efficient Foundation Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2302.13971","snapshot_observed_at":"2026-08-16T10:28:52.512486Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2504.18154","last_updated":"2025-04-25T08:06:22Z","snapshot_observed_at":"2026-08-18T08:38:54.349344Z","submitted_at":"2025-04-25T08:06:22Z","title":"EcoServe: Enabling Cost-effective LLM Serving with Proactive Intra- and Inter-Instance Orchestration","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-16T10:28:52.512486Z"},"links":{"cited_paper":"/paper/2302.13971","citing_paper":"/paper/2504.18154"},"observation_digest":"sha256:e59a5f6f24d0c4ea6110455a8bc549a92c54ebac09ee43f750fe04204c4cddf2","observation_id":"3e56e867-a6af-4dae-abed-6347f9733b08","resolution":{"observed_at":"2026-08-16T10:28:52.512486Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:28:52.516329Z","title":null,"venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2504.18154","last_updated":"2025-04-25T08:06:22Z","snapshot_observed_at":"2026-08-18T08:38:54.349344Z","submitted_at":"2025-04-25T08:06:22Z","title":"EcoServe: Enabling Cost-effective LLM Serving with Proactive Intra- and Inter-Instance Orchestration","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-16T10:28:52.516329Z"},"links":{"citing_paper":"/paper/2504.18154"},"observation_digest":"sha256:390b22efea4411ef698ca8a8617bd7a0d4951b01a2883f986b3bf32eb31bc2ae","observation_id":"340b5c5d-7843-48c0-9201-325952e0be67","resolution":{"observed_at":"2026-08-16T10:28:52.516329Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.21465","last_updated":"2025-04-25T19:40:54Z","snapshot_observed_at":"2026-08-16T13:05:14.451211Z","submitted_at":"2024-10-28T19:08:12Z","title":"ShadowKV: KV Cache in Shadows for High-Throughput Long-Context LLM Inference","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.21465","snapshot_observed_at":"2026-08-16T10:28:52.508979Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2504.18154","last_updated":"2025-04-25T08:06:22Z","snapshot_observed_at":"2026-08-18T08:38:54.349344Z","submitted_at":"2025-04-25T08:06:22Z","title":"EcoServe: Enabling Cost-effective LLM Serving with Proactive Intra- and Inter-Instance Orchestration","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-16T10:28:52.508979Z"},"links":{"cited_paper":"/paper/2410.21465","citing_paper":"/paper/2504.18154"},"observation_digest":"sha256:6342ba74c80aa0be593169ee91b8d84ca5dd27c79da8c115b582a5629b8ef58e","observation_id":"d19f214f-d719-492e-83ff-27baf0ba7ae9","resolution":{"observed_at":"2026-08-16T10:28:52.508979Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.10671","last_updated":"2024-09-10T13:25:53Z","snapshot_observed_at":"2026-08-17T11:08:48.802438Z","submitted_at":"2024-07-15T12:35:42Z","title":"Qwen2 Technical Report","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.10671","snapshot_observed_at":"2026-08-16T10:28:52.523540Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2504.18154","last_updated":"2025-04-25T08:06:22Z","snapshot_observed_at":"2026-08-18T08:38:54.349344Z","submitted_at":"2025-04-25T08:06:22Z","title":"EcoServe: Enabling Cost-effective LLM Serving with Proactive Intra- and Inter-Instance Orchestration","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-16T10:28:52.523540Z"},"links":{"cited_paper":"/paper/2407.10671","citing_paper":"/paper/2504.18154"},"observation_digest":"sha256:22e07ef0105462894b49550048c7df78b585918fce3cf80bac73aad8cc5b7589","observation_id":"3412e589-6bb7-447f-993c-78c4968a6340","resolution":{"observed_at":"2026-08-16T10:28:52.523540Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:28:52.527545Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2504.18154","last_updated":"2025-04-25T08:06:22Z","snapshot_observed_at":"2026-08-18T08:38:54.349344Z","submitted_at":"2025-04-25T08:06:22Z","title":"EcoServe: Enabling Cost-effective LLM Serving with Proactive Intra- and Inter-Instance Orchestration","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-16T10:28:52.527545Z"},"links":{"citing_paper":"/paper/2504.18154"},"observation_digest":"sha256:572de00496df74bc38014a17a782ebb41cf6b221d343b8119cf1eaff24b0c739","observation_id":"61225a82-7ee6-43b7-b7ff-3bd027395a13","resolution":{"observed_at":"2026-08-16T10:28:52.527545Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:28:52.780452Z","title":null,"venue":null,"work_id":"a66d193d-64c9-4aca-b204-8b7ff3dbcbbf","year":2024},"citing_paper":{"arxiv_id":"2504.18154","last_updated":"2025-04-25T08:06:22Z","snapshot_observed_at":"2026-08-18T08:38:54.349344Z","submitted_at":"2025-04-25T08:06:22Z","title":"EcoServe: Enabling Cost-effective LLM Serving with Proactive Intra- and Inter-Instance Orchestration","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-16T10:28:52.520287Z"},"links":{"citing_paper":"/paper/2504.18154"},"observation_digest":"sha256:2b8014ce8c70ac188ccf3fcaa067b6fdfb8a12e2c43f224413696ff4541e51ff","observation_id":"38ee51ee-5e49-42de-9c2d-f94d75adb60e","resolution":{"observed_at":"2026-08-16T10:28:52.784003Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:28:52.534148Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2504.18154","last_updated":"2025-04-25T08:06:22Z","snapshot_observed_at":"2026-08-18T08:38:54.349344Z","submitted_at":"2025-04-25T08:06:22Z","title":"EcoServe: Enabling Cost-effective LLM Serving with Proactive Intra- and Inter-Instance Orchestration","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-16T10:28:52.534148Z"},"links":{"citing_paper":"/paper/2504.18154"},"observation_digest":"sha256:cad87629ec16f21bf2f3a641760f785094b7099453cbaca1d7c2d7c15f17c6c2","observation_id":"14986e58-8f2a-4aa0-bca2-c52300b3fac4","resolution":{"observed_at":"2026-08-16T10:28:52.534148Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:28:52.746001Z","title":"2024.{DistServe}: Disaggregating Prefill and Decoding for Goodput-optimized Large Language Model Serving","venue":null,"work_id":"778201bf-55b1-4911-84a6-014d637e1915","year":2024},"citing_paper":{"arxiv_id":"2504.18154","last_updated":"2025-04-25T08:06:22Z","snapshot_observed_at":"2026-08-18T08:38:54.349344Z","submitted_at":"2025-04-25T08:06:22Z","title":"EcoServe: Enabling Cost-effective LLM Serving with Proactive Intra- and Inter-Instance Orchestration","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-16T10:28:52.537827Z"},"links":{"citing_paper":"/paper/2504.18154"},"observation_digest":"sha256:c92f7cd316ce5a01e4393b2fea5b7cd471afc56157b156eee307c7bdba9d8418","observation_id":"74e009ea-b3cc-421a-815d-4dae67bb665f","resolution":{"observed_at":"2026-08-16T10:28:52.750772Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:28:52.763262Z","title":null,"venue":null,"work_id":"09a7bc77-3dd1-48c6-8833-132966fd2dad","year":2022},"citing_paper":{"arxiv_id":"2504.18154","last_updated":"2025-04-25T08:06:22Z","snapshot_observed_at":"2026-08-18T08:38:54.349344Z","submitted_at":"2025-04-25T08:06:22Z","title":"EcoServe: Enabling Cost-effective LLM Serving with Proactive Intra- and Inter-Instance Orchestration","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-16T10:28:52.531054Z"},"links":{"citing_paper":"/paper/2504.18154"},"observation_digest":"sha256:10df627ae6de96fb32204ca3acc2d7e3fe4e37592640f2c1e43bc91ef06c70a3","observation_id":"38676f8d-7ba9-4aed-940a-ffd7b138c995","resolution":{"observed_at":"2026-08-16T10:28:52.766731Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.02263","last_updated":"2025-07-26T15:29:10Z","snapshot_observed_at":"2026-08-16T12:44:24.037349Z","submitted_at":"2025-04-03T04:20:44Z","title":"MegaScale-Infer: Serving Mixture-of-Experts at Scale with Disaggregated Expert Parallelism","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.02263","snapshot_observed_at":"2026-08-16T10:28:52.545392Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2504.18154","last_updated":"2025-04-25T08:06:22Z","snapshot_observed_at":"2026-08-18T08:38:54.349344Z","submitted_at":"2025-04-25T08:06:22Z","title":"EcoServe: Enabling Cost-effective LLM Serving with Proactive Intra- and Inter-Instance Orchestration","version":1},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-16T10:28:52.545392Z"},"links":{"cited_paper":"/paper/2504.02263","citing_paper":"/paper/2504.18154"},"observation_digest":"sha256:29803a94ff6645a0ce55ff9fef1568eac94e77ebec1e29690ec1b5fb7e6a21b9","observation_id":"d148bb34-cbcb-481e-ba96-560cfb75257d","resolution":{"observed_at":"2026-08-16T10:28:52.545392Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.12757","last_updated":"2025-05-25T14:08:01Z","snapshot_observed_at":"2026-08-17T17:38:19.809822Z","submitted_at":"2024-08-22T23:00:40Z","title":"NanoFlow: Towards Optimal Large Language Model Serving Throughput","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.12757","snapshot_observed_at":"2026-08-16T10:28:52.541781Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2504.18154","last_updated":"2025-04-25T08:06:22Z","snapshot_observed_at":"2026-08-18T08:38:54.349344Z","submitted_at":"2025-04-25T08:06:22Z","title":"EcoServe: Enabling Cost-effective LLM Serving with Proactive Intra- and Inter-Instance Orchestration","version":1},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-08-16T10:28:52.541781Z"},"links":{"cited_paper":"/paper/2408.12757","citing_paper":"/paper/2504.18154"},"observation_digest":"sha256:4f135ee99ca77010f7c1310f3928332d04dcaf24a98deeebfd8ca7b527bb2710","observation_id":"a72c7eef-a6da-43ce-8259-320b0f05cb75","resolution":{"observed_at":"2026-08-16T10:28:52.541781Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:28:52.400146Z","title":"Advances in neural information processing systems 35 (2022), 16344–16359","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2504.18154","last_updated":"2025-04-25T08:06:22Z","snapshot_observed_at":"2026-08-18T08:38:54.349344Z","submitted_at":"2025-04-25T08:06:22Z","title":"EcoServe: Enabling Cost-effective LLM Serving with Proactive Intra- and Inter-Instance Orchestration","version":1},"reference_index":2022,"source":"pdf_text","source_observed_at":"2026-08-16T10:28:52.400146Z"},"links":{"citing_paper":"/paper/2504.18154"},"observation_digest":"sha256:f50f00a0294d63ac2f4a2d2abfa15231cccefc3664b5e09401eb4a1d3ae51055","observation_id":"f5c06792-eb29-4484-a705-8ed01eded5bf","resolution":{"observed_at":"2026-08-16T10:28:52.400146Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-16T10:28:52.439069Z","title":"In Proceedings of the 29th Symposium on Operating Systems Principles","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2504.18154","last_updated":"2025-04-25T08:06:22Z","snapshot_observed_at":"2026-08-18T08:38:54.349344Z","submitted_at":"2025-04-25T08:06:22Z","title":"EcoServe: Enabling Cost-effective LLM Serving with Proactive Intra- and Inter-Instance Orchestration","version":1},"reference_index":2023,"source":"pdf_text","source_observed_at":"2026-08-16T10:28:52.439069Z"},"links":{"citing_paper":"/paper/2504.18154"},"observation_digest":"sha256:d8c6cf48dfc566e7d22865130b4df28c37751c715da6bd46e147c1afe238919a","observation_id":"a64b634a-0847-4491-83e8-2d4cd932fee9","resolution":{"observed_at":"2026-08-16T10:28:52.439069Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2504.18154","last_updated":"2025-04-25T08:06:22Z","latest_version":1,"primary_category":"cs.DC","snapshot_observed_at":"2026-08-18T08:38:54.349344Z","submitted_at":"2025-04-25T08:06:22Z","title":"EcoServe: Enabling Cost-effective LLM Serving with Proactive Intra- and Inter-Instance Orchestration"},"reference_resolution":{"displayed":55,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":50,"verified_exact":0,"verified_fuzzy":5},"total_outbound_references":55},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"thesis":"As of 18 August 2026, this Paper Citation Record lists 55 of 55 outbound references and 2 inbound Pith citation observations for arXiv:2504.18154."}