{"as_of":"2026-08-08T04:50:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:d871ee7f1bfdb7dbb77f1a9be0d636540178bfeb59f14d7b2c2026b84272f453","coverage":[{"denominator":36,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":36,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T10:18:24.943745Z","state":"measured"},{"denominator":37,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":37,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-07T06:34:17.273281+00:00","state":"measured"},{"denominator":1,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":1,"source":"paper_references, paper_reference_links","source_observed_at":"2026-07-14T20:35:48.828322Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2506.05871","last_updated":"2025-06-06T08:40:10Z","snapshot_observed_at":"2026-08-07T23:13:18.541096Z","submitted_at":"2025-06-06T08:40:10Z","title":"BestServe: Serving Strategies with Optimal Goodput in Collocation and Disaggregation Architectures","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2506.05871","snapshot_observed_at":"2026-07-14T20:35:48.828322Z","title":"Bestserve: Serving strategies with optimal good- put in collocation and disaggregation architectures.CoRR abs/2506.05871(2025)","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2603.15202","last_updated":"2026-06-05T04:55:52Z","snapshot_observed_at":"2026-07-14T20:35:46.715134Z","submitted_at":"2026-03-16T12:43:32Z","title":"Simple is Better: Multiplication May Be All You Need for LLM Request Scheduling","version":3},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-07-14T20:35:48.828322Z"},"links":{"cited_paper":"/paper/2506.05871","citing_paper":"/paper/2603.15202"},"observation_digest":"sha256:dca8cffe4f3801c62bfa059c49e5c9344fa5a161288df6fd0be29d6ac0f76d93","observation_id":"23e4ae32-5d5a-4207-aa2e-5810f04e42cb","resolution":{"observed_at":"2026-07-14T20:35:48.828322Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2506.05871/citation-record","integrity":"/paper/2506.05871/integrity","json":"/paper/2506.05871/citation-record.json","paper":"/paper/2506.05871"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:18:21.975462Z","title":"Gqa: Training generalized multi-query transformer models from multi-head checkpoints, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.05871","last_updated":"2025-06-06T08:40:10Z","snapshot_observed_at":"2026-08-07T23:13:18.541096Z","submitted_at":"2025-06-06T08:40:10Z","title":"BestServe: Serving Strategies with Optimal Goodput in Collocation and Disaggregation Architectures","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-07T10:18:21.975462Z"},"links":{"citing_paper":"/paper/2506.05871"},"observation_digest":"sha256:ec30201a6101d04e81384d8ea17cf1e3ea4e23bb9b597c312d93ff4978fc9841","observation_id":"9a84f048-fc01-47c7-bdd5-71fb2a72c78c","resolution":{"observed_at":"2026-08-07T10:18:21.975462Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:18:31.975345Z","title":"How continuous batching enables 23x throughput in LLM inference while reducing p50 latency","venue":null,"work_id":"e921f029-d308-45b9-88f7-3097058eb653","year":2023},"citing_paper":{"arxiv_id":"2506.05871","last_updated":"2025-06-06T08:40:10Z","snapshot_observed_at":"2026-08-07T23:13:18.541096Z","submitted_at":"2025-06-06T08:40:10Z","title":"BestServe: Serving Strategies with Optimal Goodput in Collocation and Disaggregation Architectures","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-07T10:18:22.047032Z"},"links":{"citing_paper":"/paper/2506.05871"},"observation_digest":"sha256:2ba97043cf4593133dcf194235d1701ddc4c87a6eb3bafb3e4103a6a9f61c6b3","observation_id":"b76c699d-04ab-4a89-9df4-bb88f9d7a139","resolution":{"observed_at":"2026-08-07T10:18:32.117041Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:18:31.763102Z","title":null,"venue":null,"work_id":"f7f55a79-8551-4dd7-a253-b1ff037f4a97","year":null},"citing_paper":{"arxiv_id":"2506.05871","last_updated":"2025-06-06T08:40:10Z","snapshot_observed_at":"2026-08-07T23:13:18.541096Z","submitted_at":"2025-06-06T08:40:10Z","title":"BestServe: Serving Strategies with Optimal Goodput in Collocation and Disaggregation Architectures","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-07T10:18:22.111097Z"},"links":{"citing_paper":"/paper/2506.05871"},"observation_digest":"sha256:4f022d9b3809e8a93e61de62e5c7dabb91b3def7c546000701c8698e38e5d4e9","observation_id":"f116b98f-bafb-43fe-a337-bb60bf1d77e5","resolution":{"observed_at":"2026-08-07T10:18:31.863421Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:18:22.166883Z","title":null,"venue":null,"work_id":null,"year":2016},"citing_paper":{"arxiv_id":"2506.05871","last_updated":"2025-06-06T08:40:10Z","snapshot_observed_at":"2026-08-07T23:13:18.541096Z","submitted_at":"2025-06-06T08:40:10Z","title":"BestServe: Serving Strategies with Optimal Goodput in Collocation and Disaggregation Architectures","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-07T10:18:22.166883Z"},"links":{"citing_paper":"/paper/2506.05871"},"observation_digest":"sha256:12f6e3486db99d3bd51f9264705e9c0b97a96a4363200b18abb2f7c0be338ad1","observation_id":"d3569a65-42ef-4074-a853-8884f11aae61","resolution":{"observed_at":"2026-08-07T10:18:22.166883Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:18:31.463964Z","title":"Throughput is not all you need: Maximizing goodput in llm serving using prefill-decode disaggregation.https://hao-ai- lab.github.io/blogs/distserve/, 2024","venue":null,"work_id":"c056b4f8-aec4-4a3d-bb98-f98df993c232","year":2024},"citing_paper":{"arxiv_id":"2506.05871","last_updated":"2025-06-06T08:40:10Z","snapshot_observed_at":"2026-08-07T23:13:18.541096Z","submitted_at":"2025-06-06T08:40:10Z","title":"BestServe: Serving Strategies with Optimal Goodput in Collocation and Disaggregation Architectures","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-07T10:18:22.238130Z"},"links":{"citing_paper":"/paper/2506.05871"},"observation_digest":"sha256:4099f4731564ba767d9c5069443af0ccb0b499e4784baec4e6c14933d6f5baf8","observation_id":"4ffec0d6-76ce-4efd-934c-30698fb7ab4c","resolution":{"observed_at":"2026-08-07T10:18:31.602003Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:18:31.207253Z","title":null,"venue":null,"work_id":"52506093-6cee-485d-842c-e2103a3d0028","year":null},"citing_paper":{"arxiv_id":"2506.05871","last_updated":"2025-06-06T08:40:10Z","snapshot_observed_at":"2026-08-07T23:13:18.541096Z","submitted_at":"2025-06-06T08:40:10Z","title":"BestServe: Serving Strategies with Optimal Goodput in Collocation and Disaggregation Architectures","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-07T10:18:22.310379Z"},"links":{"citing_paper":"/paper/2506.05871"},"observation_digest":"sha256:73446d231fb7dba7a49009b9cb741a2a29a4c270db06f944e44af72575a7f828","observation_id":"3a7fe4fb-4914-4440-b118-d8163dfad57a","resolution":{"observed_at":"2026-08-07T10:18:31.356513Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:18:30.943286Z","title":null,"venue":null,"work_id":"79073874-02f6-498f-99e5-9d36bdc27251","year":null},"citing_paper":{"arxiv_id":"2506.05871","last_updated":"2025-06-06T08:40:10Z","snapshot_observed_at":"2026-08-07T23:13:18.541096Z","submitted_at":"2025-06-06T08:40:10Z","title":"BestServe: Serving Strategies with Optimal Goodput in Collocation and Disaggregation Architectures","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-07T10:18:22.376684Z"},"links":{"citing_paper":"/paper/2506.05871"},"observation_digest":"sha256:8c5d88fc91ddb4b00f34809f7966e1ed448c2c29d57b65ded565738af4125f2b","observation_id":"f8c79ad7-811b-472b-ab2e-bb55a4b7f564","resolution":{"observed_at":"2026-08-07T10:18:31.052695Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:18:30.722727Z","title":null,"venue":null,"work_id":"6e140df6-8ce0-429a-82df-c1d948da4913","year":2022},"citing_paper":{"arxiv_id":"2506.05871","last_updated":"2025-06-06T08:40:10Z","snapshot_observed_at":"2026-08-07T23:13:18.541096Z","submitted_at":"2025-06-06T08:40:10Z","title":"BestServe: Serving Strategies with Optimal Goodput in Collocation and Disaggregation Architectures","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-07T10:18:22.458150Z"},"links":{"citing_paper":"/paper/2506.05871"},"observation_digest":"sha256:1272d79e074b3c6d6f84bee73b67b2c2bbfd65da2954fe269a43c49937e70b05","observation_id":"32076b1c-5f97-41ab-ab8d-614d6ea9393c","resolution":{"observed_at":"2026-08-07T10:18:30.817164Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:18:22.540447Z","title":"Sigmoid-weighted linear units for neural network function approximation in reinforcement learning, 2017","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2506.05871","last_updated":"2025-06-06T08:40:10Z","snapshot_observed_at":"2026-08-07T23:13:18.541096Z","submitted_at":"2025-06-06T08:40:10Z","title":"BestServe: Serving Strategies with Optimal Goodput in Collocation and Disaggregation Architectures","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-07T10:18:22.540447Z"},"links":{"citing_paper":"/paper/2506.05871"},"observation_digest":"sha256:7cef0cabc6a0e575d1f8b689d8664486dcd3df95d657a529f15c71ef8b47c3f0","observation_id":"24b96557-fb56-44a3-95ff-7b54f2e5471a","resolution":{"observed_at":"2026-08-07T10:18:22.540447Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:18:30.467478Z","title":"Text generation inference","venue":null,"work_id":"9745a484-91cc-4d51-b4ba-ff0bb2ee7b78","year":null},"citing_paper":{"arxiv_id":"2506.05871","last_updated":"2025-06-06T08:40:10Z","snapshot_observed_at":"2026-08-07T23:13:18.541096Z","submitted_at":"2025-06-06T08:40:10Z","title":"BestServe: Serving Strategies with Optimal Goodput in Collocation and Disaggregation Architectures","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-07T10:18:22.605821Z"},"links":{"citing_paper":"/paper/2506.05871"},"observation_digest":"sha256:4c0ccd250bf496a19b7554f89591721d82d3d428b2cc7c1b4d25d1e1b577f685","observation_id":"c16b42c3-e6aa-402a-918d-71d2ccaafcb1","resolution":{"observed_at":"2026-08-07T10:18:30.599741Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:18:30.157260Z","title":"Low latency rnn inference with cellular batching","venue":null,"work_id":"6d048390-5939-40a9-8fc2-48104d4ca736","year":2018},"citing_paper":{"arxiv_id":"2506.05871","last_updated":"2025-06-06T08:40:10Z","snapshot_observed_at":"2026-08-07T23:13:18.541096Z","submitted_at":"2025-06-06T08:40:10Z","title":"BestServe: Serving Strategies with Optimal Goodput in Collocation and Disaggregation Architectures","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-07T10:18:22.666922Z"},"links":{"citing_paper":"/paper/2506.05871"},"observation_digest":"sha256:844a949142db9993582e1e6e477966fe6eaaeffc9efc939462b485666fa63553","observation_id":"6e795329-54e9-4350-9bd8-073f3c64a412","resolution":{"observed_at":"2026-08-07T10:18:30.273233Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:18:29.924463Z","title":"Getting started with CUDA graphs","venue":null,"work_id":"5d585c26-1361-4674-92c5-0848d8620754","year":2019},"citing_paper":{"arxiv_id":"2506.05871","last_updated":"2025-06-06T08:40:10Z","snapshot_observed_at":"2026-08-07T23:13:18.541096Z","submitted_at":"2025-06-06T08:40:10Z","title":"BestServe: Serving Strategies with Optimal Goodput in Collocation and Disaggregation Architectures","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-07T10:18:22.721275Z"},"links":{"citing_paper":"/paper/2506.05871"},"observation_digest":"sha256:e66f4e86175052435126fd04bf96d43817bf3053c2c7d0aa401ba07bde307910","observation_id":"d676dfcc-085d-457a-8234-0659cbf90478","resolution":{"observed_at":"2026-08-07T10:18:30.035124Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:18:29.606295Z","title":"Shortle, James M","venue":null,"work_id":"173ffa6a-08bb-4833-a8fd-ca9b9e5e6535","year":2008},"citing_paper":{"arxiv_id":"2506.05871","last_updated":"2025-06-06T08:40:10Z","snapshot_observed_at":"2026-08-07T23:13:18.541096Z","submitted_at":"2025-06-06T08:40:10Z","title":"BestServe: Serving Strategies with Optimal Goodput in Collocation and Disaggregation Architectures","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-07T10:18:22.782207Z"},"links":{"citing_paper":"/paper/2506.05871"},"observation_digest":"sha256:d50a3cd5d54fe9dec3a64fe7ab0e9edb1861b241189eb833eca3f492f8bfc0af","observation_id":"86bb2b9d-0fd9-496c-b9fb-290bcd0678d7","resolution":{"observed_at":"2026-08-07T10:18:29.734840Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:18:29.359822Z","title":"Pipedream: Fast and efficient pipeline parallel dnn training, 2018","venue":null,"work_id":"2c89ca97-367b-4712-8488-44da797743a4","year":2018},"citing_paper":{"arxiv_id":"2506.05871","last_updated":"2025-06-06T08:40:10Z","snapshot_observed_at":"2026-08-07T23:13:18.541096Z","submitted_at":"2025-06-06T08:40:10Z","title":"BestServe: Serving Strategies with Optimal Goodput in Collocation and Disaggregation Architectures","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-07T10:18:22.842703Z"},"links":{"citing_paper":"/paper/2506.05871"},"observation_digest":"sha256:63873de6743e7b301cff2de5a9dfbba1f23ec5a4afbbd2f3adf92025a081c210","observation_id":"0aee096f-153b-446d-b9b7-33c76b19bbb1","resolution":{"observed_at":"2026-08-07T10:18:29.478952Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:18:22.918596Z","title":"Inference without interference: Disaggregate llm inference for mixed downstream workloads, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.05871","last_updated":"2025-06-06T08:40:10Z","snapshot_observed_at":"2026-08-07T23:13:18.541096Z","submitted_at":"2025-06-06T08:40:10Z","title":"BestServe: Serving Strategies with Optimal Goodput in Collocation and Disaggregation Architectures","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-07T10:18:22.918596Z"},"links":{"citing_paper":"/paper/2506.05871"},"observation_digest":"sha256:3fbc7463681761bb7c47a4b8ed3eb4b2ba97ca0b2538278363010dbf4238bdf9","observation_id":"f32f8b3b-45e4-4fcc-8ad3-536d46c7d48a","resolution":{"observed_at":"2026-08-07T10:18:22.918596Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:18:29.117816Z","title":"Le, Yonghui Wu, and Zhifeng Chen.GPipe: efficient training of giant neural networks using pipeline parallelism","venue":null,"work_id":"c86c6722-1c14-4c41-bdb0-72d42a5a71b5","year":2019},"citing_paper":{"arxiv_id":"2506.05871","last_updated":"2025-06-06T08:40:10Z","snapshot_observed_at":"2026-08-07T23:13:18.541096Z","submitted_at":"2025-06-06T08:40:10Z","title":"BestServe: Serving Strategies with Optimal Goodput in Collocation and Disaggregation Architectures","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-07T10:18:23.004203Z"},"links":{"citing_paper":"/paper/2506.05871"},"observation_digest":"sha256:373aa987aca41ecbcaf535ec1aef691853105e0639973baa502ea67b422fa44e","observation_id":"fc0b7dfd-3473-4204-aaa8-8d8ad4ed900b","resolution":{"observed_at":"2026-08-07T10:18:29.186315Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:18:23.064579Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.05871","last_updated":"2025-06-06T08:40:10Z","snapshot_observed_at":"2026-08-07T23:13:18.541096Z","submitted_at":"2025-06-06T08:40:10Z","title":"BestServe: Serving Strategies with Optimal Goodput in Collocation and Disaggregation Architectures","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-07T10:18:23.064579Z"},"links":{"citing_paper":"/paper/2506.05871"},"observation_digest":"sha256:188743a3d48782544d330c5fefc4ef5f7208f7e5736ddaf6b975fb1cce278f65","observation_id":"6a9ad0c6-582b-4c8c-bf15-b812ea35e28e","resolution":{"observed_at":"2026-08-07T10:18:23.064579Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:18:29.064604Z","title":"Efficient memory management for large language model serving with PagedAttention","venue":null,"work_id":"cf6e0603-334f-413a-851d-ed43c20fed44","year":2023},"citing_paper":{"arxiv_id":"2506.05871","last_updated":"2025-06-06T08:40:10Z","snapshot_observed_at":"2026-08-07T23:13:18.541096Z","submitted_at":"2025-06-06T08:40:10Z","title":"BestServe: Serving Strategies with Optimal Goodput in Collocation and Disaggregation Architectures","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-07T10:18:23.122706Z"},"links":{"citing_paper":"/paper/2506.05871"},"observation_digest":"sha256:c182537c212f5ae12ab1b0de628cdf7af77a820a80ffbcc328780bf8e80a74d0","observation_id":"145f5443-f828-4631-87b0-8ea6a3de16bb","resolution":{"observed_at":"2026-08-07T10:18:29.105821Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:18:28.879122Z","title":"Transformers KV caching explained","venue":null,"work_id":"e5d5ae6b-5ac9-4d8a-abcf-b0d37f56b6c3","year":2023},"citing_paper":{"arxiv_id":"2506.05871","last_updated":"2025-06-06T08:40:10Z","snapshot_observed_at":"2026-08-07T23:13:18.541096Z","submitted_at":"2025-06-06T08:40:10Z","title":"BestServe: Serving Strategies with Optimal Goodput in Collocation and Disaggregation Architectures","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-07T10:18:23.182303Z"},"links":{"citing_paper":"/paper/2506.05871"},"observation_digest":"sha256:711efb9da38cfe0ee9792ea2b500413db113a1b37796058a7b12b234346cfbe7","observation_id":"0d23d8a8-73ad-40d6-91da-9b8aedc2cdec","resolution":{"observed_at":"2026-08-07T10:18:29.006858Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:18:28.660771Z","title":"Sequence parallelism: Long sequence training from system perspective, 2022","venue":null,"work_id":"a894bff3-67e6-4c76-b8ab-95f7dbc7e4ee","year":2022},"citing_paper":{"arxiv_id":"2506.05871","last_updated":"2025-06-06T08:40:10Z","snapshot_observed_at":"2026-08-07T23:13:18.541096Z","submitted_at":"2025-06-06T08:40:10Z","title":"BestServe: Serving Strategies with Optimal Goodput in Collocation and Disaggregation Architectures","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-07T10:18:23.303096Z"},"links":{"citing_paper":"/paper/2506.05871"},"observation_digest":"sha256:f931df5a4062b8773bd35c628797eee753f60457adf284468338119d396d31db","observation_id":"45caf8d9-9714-4566-b9fa-0db1dcdb3cea","resolution":{"observed_at":"2026-08-07T10:18:28.761856Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:18:28.426744Z","title":"Llama 3.2: Revolutionizing edge AI and vision with open, customizable models","venue":null,"work_id":"e4329bd4-ec3a-4e1f-b51a-c09eb8805fe0","year":2024},"citing_paper":{"arxiv_id":"2506.05871","last_updated":"2025-06-06T08:40:10Z","snapshot_observed_at":"2026-08-07T23:13:18.541096Z","submitted_at":"2025-06-06T08:40:10Z","title":"BestServe: Serving Strategies with Optimal Goodput in Collocation and Disaggregation Architectures","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-07T10:18:23.353733Z"},"links":{"citing_paper":"/paper/2506.05871"},"observation_digest":"sha256:6371e2483cbf76086934eb986c053ed850b5d6c5261ae514624ec2903c30e506","observation_id":"64e64901-3b20-4764-b6a6-651291e110b3","resolution":{"observed_at":"2026-08-07T10:18:28.549453Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:18:28.230288Z","title":"NVIDIA TensorRT-LLM.https: //docs.nvidia.com/tensorrt-llm/index.html","venue":null,"work_id":"20b642e6-11b6-4d92-a1f6-b37e0eb574c4","year":null},"citing_paper":{"arxiv_id":"2506.05871","last_updated":"2025-06-06T08:40:10Z","snapshot_observed_at":"2026-08-07T23:13:18.541096Z","submitted_at":"2025-06-06T08:40:10Z","title":"BestServe: Serving Strategies with Optimal Goodput in Collocation and Disaggregation Architectures","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-07T10:18:23.468448Z"},"links":{"citing_paper":"/paper/2506.05871"},"observation_digest":"sha256:4b0274aa7ea7b069c9e4968efc440aa65ecec5d3faaff8f5d8d84aaefeced4af","observation_id":"e8618f5a-0dd0-46b8-a1b4-6c6368e7b5cd","resolution":{"observed_at":"2026-08-07T10:18:28.335252Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:18:27.955936Z","title":"OpenAI o3-mini","venue":null,"work_id":"d9ffedad-bf4d-4e80-8203-476cda711da3","year":2025},"citing_paper":{"arxiv_id":"2506.05871","last_updated":"2025-06-06T08:40:10Z","snapshot_observed_at":"2026-08-07T23:13:18.541096Z","submitted_at":"2025-06-06T08:40:10Z","title":"BestServe: Serving Strategies with Optimal Goodput in Collocation and Disaggregation Architectures","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-07T10:18:23.581221Z"},"links":{"citing_paper":"/paper/2506.05871"},"observation_digest":"sha256:6b57b716029b372f4d987e0f6dae898479cca9514085a2e787a8c174b7c42ec7","observation_id":"bc6c61ce-9f85-444a-8f49-441f6dd4c07c","resolution":{"observed_at":"2026-08-07T10:18:28.084574Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:18:27.702753Z","title":"Splitwise: Efficient generative llm inference using phase splitting, 2024","venue":null,"work_id":"ad471856-02b9-4a5a-9825-ea93aeb269b5","year":2024},"citing_paper":{"arxiv_id":"2506.05871","last_updated":"2025-06-06T08:40:10Z","snapshot_observed_at":"2026-08-07T23:13:18.541096Z","submitted_at":"2025-06-06T08:40:10Z","title":"BestServe: Serving Strategies with Optimal Goodput in Collocation and Disaggregation Architectures","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-07T10:18:23.701320Z"},"links":{"citing_paper":"/paper/2506.05871"},"observation_digest":"sha256:50b71df87bbc97c08a4e5bb0942333cd53dfe5a2a83957206cf069cf9e9c3753","observation_id":"294d1086-4bb9-4ba4-8247-257ff3a68a55","resolution":{"observed_at":"2026-08-07T10:18:27.810127Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:18:27.511310Z","title":"Mooncake: A KVCache-centric disaggregated architecture for LLM serving, 2024","venue":null,"work_id":"dfcc1a96-1c52-43f1-8323-de3dd9ff5c35","year":2024},"citing_paper":{"arxiv_id":"2506.05871","last_updated":"2025-06-06T08:40:10Z","snapshot_observed_at":"2026-08-07T23:13:18.541096Z","submitted_at":"2025-06-06T08:40:10Z","title":"BestServe: Serving Strategies with Optimal Goodput in Collocation and Disaggregation Architectures","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-07T10:18:23.849100Z"},"links":{"citing_paper":"/paper/2506.05871"},"observation_digest":"sha256:2ed6badd91376850664c1279ca95af271367aa82bbe9379fd7521f0fae0e7d16","observation_id":"dd690d7d-cf99-41df-8990-1f63a9f18590","resolution":{"observed_at":"2026-08-07T10:18:27.600512Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:18:27.297862Z","title":"Focus: For tech giants, AI like Bing and Bard poses billion-dollar search problem","venue":null,"work_id":"c7321eb3-94e7-4fdd-8026-f213a83e6a9e","year":2023},"citing_paper":{"arxiv_id":"2506.05871","last_updated":"2025-06-06T08:40:10Z","snapshot_observed_at":"2026-08-07T23:13:18.541096Z","submitted_at":"2025-06-06T08:40:10Z","title":"BestServe: Serving Strategies with Optimal Goodput in Collocation and Disaggregation Architectures","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-07T10:18:23.901389Z"},"links":{"citing_paper":"/paper/2506.05871"},"observation_digest":"sha256:6c47f52d29b5a6b97c6a6547fad6bf2539e2b67b0718ccfeeee1b55f134408ba","observation_id":"fa45246a-90e3-447d-98ee-280bf8e2edb9","resolution":{"observed_at":"2026-08-07T10:18:27.412217Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:18:26.974952Z","title":null,"venue":null,"work_id":"c643c61b-72b9-4049-9da5-b191cf9f16e9","year":null},"citing_paper":{"arxiv_id":"2506.05871","last_updated":"2025-06-06T08:40:10Z","snapshot_observed_at":"2026-08-07T23:13:18.541096Z","submitted_at":"2025-06-06T08:40:10Z","title":"BestServe: Serving Strategies with Optimal Goodput in Collocation and Disaggregation Architectures","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-07T10:18:23.942676Z"},"links":{"citing_paper":"/paper/2506.05871"},"observation_digest":"sha256:66dfebae9436d5858f729774acfbf6fa62165a140109f7320b9d4a519169cd93","observation_id":"a06f4a7f-9185-4cf2-9fd8-08725cf7b59b","resolution":{"observed_at":"2026-08-07T10:18:27.145596Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:18:23.999552Z","title":"Megatron-lm: Training multi-billion parameter language models using model parallelism, 2020","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2506.05871","last_updated":"2025-06-06T08:40:10Z","snapshot_observed_at":"2026-08-07T23:13:18.541096Z","submitted_at":"2025-06-06T08:40:10Z","title":"BestServe: Serving Strategies with Optimal Goodput in Collocation and Disaggregation Architectures","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-07T10:18:23.999552Z"},"links":{"citing_paper":"/paper/2506.05871"},"observation_digest":"sha256:80aae2cc4dc80cc70241f4ec2a11f193c7eb1e21e6b2ec5953d7df6e9b4a3ddf","observation_id":"23909fa7-09e8-4b89-96af-7ccdb9dbc366","resolution":{"observed_at":"2026-08-07T10:18:23.999552Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:18:26.582950Z","title":"Gomez, Łukasz Kaiser, and Illia Polosukhin","venue":null,"work_id":"77a307ed-cfcb-48e6-9f28-713cb7826eff","year":2017},"citing_paper":{"arxiv_id":"2506.05871","last_updated":"2025-06-06T08:40:10Z","snapshot_observed_at":"2026-08-07T23:13:18.541096Z","submitted_at":"2025-06-06T08:40:10Z","title":"BestServe: Serving Strategies with Optimal Goodput in Collocation and Disaggregation Architectures","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-07T10:18:24.118959Z"},"links":{"citing_paper":"/paper/2506.05871"},"observation_digest":"sha256:f960b53b92c968359d7e51e676de02d2be8ea2711aff88884d721064460d2597","observation_id":"c7662fe8-a85d-4e52-a49c-f766c9634da4","resolution":{"observed_at":"2026-08-07T10:18:26.742702Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:18:26.279744Z","title":"https://docs.vllm.ai/en/v0.4.2/index.html","venue":null,"work_id":"94d4a7e8-f046-454c-87e6-0556b2ec68c9","year":null},"citing_paper":{"arxiv_id":"2506.05871","last_updated":"2025-06-06T08:40:10Z","snapshot_observed_at":"2026-08-07T23:13:18.541096Z","submitted_at":"2025-06-06T08:40:10Z","title":"BestServe: Serving Strategies with Optimal Goodput in Collocation and Disaggregation Architectures","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-07T10:18:24.229841Z"},"links":{"citing_paper":"/paper/2506.05871"},"observation_digest":"sha256:4db8d4500d84cc00f3f430d711c1ee02d705ef048b85ff11bb79c8448d4f0b8f","observation_id":"ed9c0fd2-6578-4842-a611-4a965f03cead","resolution":{"observed_at":"2026-08-07T10:18:26.408769Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:18:25.972872Z","title":"https://github.com/vllm-project/vllm-ascend","venue":null,"work_id":"7d3fcdf6-0ed2-4653-8d0e-fab3e3eb68d2","year":null},"citing_paper":{"arxiv_id":"2506.05871","last_updated":"2025-06-06T08:40:10Z","snapshot_observed_at":"2026-08-07T23:13:18.541096Z","submitted_at":"2025-06-06T08:40:10Z","title":"BestServe: Serving Strategies with Optimal Goodput in Collocation and Disaggregation Architectures","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-07T10:18:24.346122Z"},"links":{"citing_paper":"/paper/2506.05871"},"observation_digest":"sha256:6b2116367f10b461a2e5a20c1b8fbc903b2f56cd30fb6340e19025c913402a7f","observation_id":"2f0c0e4d-6004-450f-8805-292f75ab23b8","resolution":{"observed_at":"2026-08-07T10:18:26.114753Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:18:25.738416Z","title":"Simai: Unifying architecture design and performance tuning for large-scale large language model training with scalability and precision","venue":null,"work_id":"09de92a9-b286-4f52-b79e-9a38790141c2","year":2025},"citing_paper":{"arxiv_id":"2506.05871","last_updated":"2025-06-06T08:40:10Z","snapshot_observed_at":"2026-08-07T23:13:18.541096Z","submitted_at":"2025-06-06T08:40:10Z","title":"BestServe: Serving Strategies with Optimal Goodput in Collocation and Disaggregation Architectures","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-07T10:18:24.449900Z"},"links":{"citing_paper":"/paper/2506.05871"},"observation_digest":"sha256:3d289a7af0f14e97e39d1fc015625c51408d5579a22fb02a0c3a5f1fd74f6100","observation_id":"248364bd-e2c4-427e-ac45-4f940f6b129f","resolution":{"observed_at":"2026-08-07T10:18:25.835717Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:18:25.489671Z","title":"Roofline: an insightful visual performance model for multicore architectures.Commun","venue":null,"work_id":"4b495f4c-fab7-44ad-b9c7-aa7b6a00ca49","year":2009},"citing_paper":{"arxiv_id":"2506.05871","last_updated":"2025-06-06T08:40:10Z","snapshot_observed_at":"2026-08-07T23:13:18.541096Z","submitted_at":"2025-06-06T08:40:10Z","title":"BestServe: Serving Strategies with Optimal Goodput in Collocation and Disaggregation Architectures","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-07T10:18:24.579266Z"},"links":{"citing_paper":"/paper/2506.05871"},"observation_digest":"sha256:ab254f8961ceb1476a5fc2f3a247e5d9f510795e4a0ebdc89dc7af3374c7194e","observation_id":"86364566-a7b8-4d13-b770-c3fc07d52163","resolution":{"observed_at":"2026-08-07T10:18:25.636873Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:18:25.264935Z","title":"Orca: A distributed serving system for Transformer-Based generative models","venue":null,"work_id":"de0d95b8-bc7a-43cb-a8e7-6e9b5c7a690b","year":2022},"citing_paper":{"arxiv_id":"2506.05871","last_updated":"2025-06-06T08:40:10Z","snapshot_observed_at":"2026-08-07T23:13:18.541096Z","submitted_at":"2025-06-06T08:40:10Z","title":"BestServe: Serving Strategies with Optimal Goodput in Collocation and Disaggregation Architectures","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-07T10:18:24.712230Z"},"links":{"citing_paper":"/paper/2506.05871"},"observation_digest":"sha256:a3d5908b39ceafbf27afa5d5ae1cc7cae68fe4d20d1609bcf14ade29f997060e","observation_id":"9ac34d5f-de27-4130-a262-bbfaf2fe842d","resolution":{"observed_at":"2026-08-07T10:18:25.376580Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:18:24.846042Z","title":"Root mean square layer normalization, 2019","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2506.05871","last_updated":"2025-06-06T08:40:10Z","snapshot_observed_at":"2026-08-07T23:13:18.541096Z","submitted_at":"2025-06-06T08:40:10Z","title":"BestServe: Serving Strategies with Optimal Goodput in Collocation and Disaggregation Architectures","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-07T10:18:24.846042Z"},"links":{"citing_paper":"/paper/2506.05871"},"observation_digest":"sha256:e1c6ba45c9ba9b486c03ab7b2a137243f42b011f175e19ac2454a141238d832b","observation_id":"9e1804fc-072c-44a9-b496-b7a853b79dec","resolution":{"observed_at":"2026-08-07T10:18:24.846042Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T10:18:25.061346Z","title":"DistServe: Disaggregating prefill and decoding for goodput-optimized large language model serving","venue":null,"work_id":"f7a39ab7-36c1-4e04-9ba4-097b00d30836","year":2024},"citing_paper":{"arxiv_id":"2506.05871","last_updated":"2025-06-06T08:40:10Z","snapshot_observed_at":"2026-08-07T23:13:18.541096Z","submitted_at":"2025-06-06T08:40:10Z","title":"BestServe: Serving Strategies with Optimal Goodput in Collocation and Disaggregation Architectures","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-07T10:18:24.943745Z"},"links":{"citing_paper":"/paper/2506.05871"},"observation_digest":"sha256:03662035619d7201223936d44f845c20e10471943624e0cc620d92c447ae0c57","observation_id":"a0d35056-e58c-43c2-81eb-367868a0e7fd","resolution":{"observed_at":"2026-08-07T10:18:25.146940Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2506.05871","last_updated":"2025-06-06T08:40:10Z","latest_version":1,"primary_category":"cs.LG","snapshot_observed_at":"2026-08-07T23:13:18.541096Z","submitted_at":"2025-06-06T08:40:10Z","title":"BestServe: Serving Strategies with Optimal Goodput in Collocation and Disaggregation Architectures"},"reference_resolution":{"displayed":36,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":12,"verified_exact":0,"verified_fuzzy":24},"total_outbound_references":36},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"thesis":"As of 8 August 2026, this Paper Citation Record lists 36 of 36 outbound references and 1 inbound Pith citation observation for arXiv:2506.05871."}