{"as_of":"2026-08-07T19:35:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:0793f9967db46117b271c96349e2a5a140326b40db1058026830e896ff1471ec","coverage":[{"denominator":64,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":64,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-06T14:45:20.821661Z","state":"measured"},{"denominator":64,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":64,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-07T06:34:17.273281+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2507.18006/citation-record","integrity":"/paper/2507.18006/integrity","json":"/paper/2507.18006/citation-record.json","paper":"/paper/2507.18006"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2303.08774","last_updated":"2024-03-04T06:01:33Z","snapshot_observed_at":"2026-08-07T07:30:12.213965Z","submitted_at":"2023-03-15T17:15:04Z","title":"GPT-4 Technical Report","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.08774","snapshot_observed_at":"2026-08-06T14:45:20.562559Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.18006","last_updated":"2025-07-24T00:49:48Z","snapshot_observed_at":"2026-08-07T13:43:37.969982Z","submitted_at":"2025-07-24T00:49:48Z","title":"Unlock the Potential of Fine-grained LLM Serving via Dynamic Module Scaling","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-06T14:45:20.562559Z"},"links":{"cited_paper":"/paper/2303.08774","citing_paper":"/paper/2507.18006"},"observation_digest":"sha256:7aeef47e7061858b34bdc199eed145d8e071c7c7d445d3454ee2f26e5e85fbd2","observation_id":"c2e070f6-fd83-452e-81fd-310b7c0dd7c0","resolution":{"observed_at":"2026-08-06T14:45:20.562559Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.09288","last_updated":"2023-07-19T17:08:59Z","snapshot_observed_at":"2026-08-07T12:56:43.323460Z","submitted_at":"2023-07-18T14:31:57Z","title":"Llama 2: Open Foundation and Fine-Tuned Chat Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.09288","snapshot_observed_at":"2026-08-06T14:45:20.567366Z","title":"Llama 2: Open foundation and fine- tuned chat models.https://arxiv.org/abs/2307.09288, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.18006","last_updated":"2025-07-24T00:49:48Z","snapshot_observed_at":"2026-08-07T13:43:37.969982Z","submitted_at":"2025-07-24T00:49:48Z","title":"Unlock the Potential of Fine-grained LLM Serving via Dynamic Module Scaling","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-06T14:45:20.567366Z"},"links":{"cited_paper":"/paper/2307.09288","citing_paper":"/paper/2507.18006"},"observation_digest":"sha256:d54fa9b2425b6f2716d37e45065ea923d163b2b055728d6012edb269bc3aa3ae","observation_id":"f5dd5571-5204-4e44-b7d8-0eccd1211acf","resolution":{"observed_at":"2026-08-06T14:45:20.567366Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.19437","last_updated":"2025-02-18T17:26:38Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-12-27T04:03:16Z","title":"DeepSeek-V3 Technical Report","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.19437","snapshot_observed_at":"2026-08-06T14:45:20.571852Z","title":"Deepseek-v3 technical report.ArXiv, abs/2412.19437, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.18006","last_updated":"2025-07-24T00:49:48Z","snapshot_observed_at":"2026-08-07T13:43:37.969982Z","submitted_at":"2025-07-24T00:49:48Z","title":"Unlock the Potential of Fine-grained LLM Serving via Dynamic Module Scaling","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-06T14:45:20.571852Z"},"links":{"cited_paper":"/paper/2412.19437","citing_paper":"/paper/2507.18006"},"observation_digest":"sha256:487b5855468e7710febdb1451c78ce6bc7530051583283ec8c7fb908b4d6abad","observation_id":"603b8f94-dccd-40a7-8181-7e40f6adac5b","resolution":{"observed_at":"2026-08-06T14:45:20.571852Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2201.03514","last_updated":"2022-06-27T08:14:54Z","snapshot_observed_at":"2026-08-01T21:13:22.714773Z","submitted_at":"2022-01-10T18:17:05Z","title":"Black-Box Tuning for Language-Model-as-a-Service","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2201.03514","snapshot_observed_at":"2026-08-06T14:45:20.576394Z","title":"Black-box tuning for language-model-as-a-service.ArXiv, abs/2201.03514, 2022","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2507.18006","last_updated":"2025-07-24T00:49:48Z","snapshot_observed_at":"2026-08-07T13:43:37.969982Z","submitted_at":"2025-07-24T00:49:48Z","title":"Unlock the Potential of Fine-grained LLM Serving via Dynamic Module Scaling","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-06T14:45:20.576394Z"},"links":{"cited_paper":"/paper/2201.03514","citing_paper":"/paper/2507.18006"},"observation_digest":"sha256:449c9829f1f6ed58c006b8a697bfcaf565b6d245f39a5a7e4eb313165990609d","observation_id":"738dbaeb-a711-43db-bfa1-62f073dc36c5","resolution":{"observed_at":"2026-08-06T14:45:20.576394Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:45:28.455276Z","title":"Xing, Hao Zhang, Joseph E","venue":null,"work_id":"bddc8d08-99fc-476c-adb6-9a82cbfc7efd","year":2023},"citing_paper":{"arxiv_id":"2507.18006","last_updated":"2025-07-24T00:49:48Z","snapshot_observed_at":"2026-08-07T13:43:37.969982Z","submitted_at":"2025-07-24T00:49:48Z","title":"Unlock the Potential of Fine-grained LLM Serving via Dynamic Module Scaling","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-06T14:45:20.581123Z"},"links":{"citing_paper":"/paper/2507.18006"},"observation_digest":"sha256:15de312cb3f51847ae78c79de8b4816fc292a03e642e51497f2b117477f1678e","observation_id":"db8eb68e-5004-45fc-b8fa-c030f11af1f7","resolution":{"observed_at":"2026-08-06T14:45:28.556477Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2303.04226","last_updated":"2023-03-07T20:36:13Z","snapshot_observed_at":"2026-07-06T14:59:53.901528Z","submitted_at":"2023-03-07T20:36:13Z","title":"A Comprehensive Survey of AI-Generated Content (AIGC): A History of Generative AI from GAN to ChatGPT","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.04226","snapshot_observed_at":"2026-08-06T14:45:20.585569Z","title":"Yu, and Lichao Sun","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.18006","last_updated":"2025-07-24T00:49:48Z","snapshot_observed_at":"2026-08-07T13:43:37.969982Z","submitted_at":"2025-07-24T00:49:48Z","title":"Unlock the Potential of Fine-grained LLM Serving via Dynamic Module Scaling","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-06T14:45:20.585569Z"},"links":{"cited_paper":"/paper/2303.04226","citing_paper":"/paper/2507.18006"},"observation_digest":"sha256:11ce254d140e499b5d082d3fac8c894352e85eb474e875de3fb24deb504dbcae","observation_id":"55b91946-2e24-42ce-8a21-b0f42ff41d2e","resolution":{"observed_at":"2026-08-06T14:45:20.585569Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2107.03374","last_updated":"2021-07-14T17:16:02Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2021-07-07T17:41:24Z","title":"Evaluating Large Language Models Trained on Code","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2107.03374","snapshot_observed_at":"2026-08-06T14:45:20.591384Z","title":"Evaluatinglarge language models trained on code.ArXiv, abs/2107.03374, 2021","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2507.18006","last_updated":"2025-07-24T00:49:48Z","snapshot_observed_at":"2026-08-07T13:43:37.969982Z","submitted_at":"2025-07-24T00:49:48Z","title":"Unlock the Potential of Fine-grained LLM Serving via Dynamic Module Scaling","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-06T14:45:20.591384Z"},"links":{"cited_paper":"/paper/2107.03374","citing_paper":"/paper/2507.18006"},"observation_digest":"sha256:88dd2647063f2a6941f18944cb1e7298fed900ca7e2243966eb5bdda4ffd3f8c","observation_id":"71922fdf-ee1e-4418-99a7-d8fa77b1303b","resolution":{"observed_at":"2026-08-06T14:45:20.591384Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.12950","last_updated":"2024-01-31T19:47:26Z","snapshot_observed_at":"2026-07-06T16:10:07.931347Z","submitted_at":"2023-08-24T17:39:13Z","title":"Code Llama: Open Foundation Models for Code","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.12950","snapshot_observed_at":"2026-08-06T14:45:20.595686Z","title":"Code llama: Open foundation models for code.ArXiv, abs/2308.12950, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.18006","last_updated":"2025-07-24T00:49:48Z","snapshot_observed_at":"2026-08-07T13:43:37.969982Z","submitted_at":"2025-07-24T00:49:48Z","title":"Unlock the Potential of Fine-grained LLM Serving via Dynamic Module Scaling","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-06T14:45:20.595686Z"},"links":{"cited_paper":"/paper/2308.12950","citing_paper":"/paper/2507.18006"},"observation_digest":"sha256:000853dc6c36bc25be99a97f29a49a30a41d6810059b891e90a20e4714cf7560","observation_id":"bdf5a873-497e-407e-862f-77e370601bbc","resolution":{"observed_at":"2026-08-06T14:45:20.595686Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:45:28.228332Z","title":"Accessed: Apr","venue":null,"work_id":"2c095341-7d4c-4ffb-96f7-a94edc02c3b5","year":2025},"citing_paper":{"arxiv_id":"2507.18006","last_updated":"2025-07-24T00:49:48Z","snapshot_observed_at":"2026-08-07T13:43:37.969982Z","submitted_at":"2025-07-24T00:49:48Z","title":"Unlock the Potential of Fine-grained LLM Serving via Dynamic Module Scaling","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-06T14:45:20.600056Z"},"links":{"citing_paper":"/paper/2507.18006"},"observation_digest":"sha256:cf346883767e139bb5a426c995a6cb9fcf5eac044923d7d23bdd1d8d908894e0","observation_id":"51fe50b3-27fa-4952-8078-3a7759ba61d3","resolution":{"observed_at":"2026-08-06T14:45:28.337931Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:45:28.084993Z","title":"Accessed: Apr","venue":null,"work_id":"e6dde3ea-5673-4789-a6d7-236fb2240ce6","year":2025},"citing_paper":{"arxiv_id":"2507.18006","last_updated":"2025-07-24T00:49:48Z","snapshot_observed_at":"2026-08-07T13:43:37.969982Z","submitted_at":"2025-07-24T00:49:48Z","title":"Unlock the Potential of Fine-grained LLM Serving via Dynamic Module Scaling","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-06T14:45:20.603833Z"},"links":{"citing_paper":"/paper/2507.18006"},"observation_digest":"sha256:727f2c30cd9c58f85fbb1cd05ae31d97eb7daa23fc226b6155fb2613076f9c32","observation_id":"b7a24f66-cd92-485f-a4af-09779ae31ac7","resolution":{"observed_at":"2026-08-06T14:45:28.135926Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:45:20.607908Z","title":"Smoothquant: Accurate and efficient post-training quantization for large language models, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.18006","last_updated":"2025-07-24T00:49:48Z","snapshot_observed_at":"2026-08-07T13:43:37.969982Z","submitted_at":"2025-07-24T00:49:48Z","title":"Unlock the Potential of Fine-grained LLM Serving via Dynamic Module Scaling","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-06T14:45:20.607908Z"},"links":{"citing_paper":"/paper/2507.18006"},"observation_digest":"sha256:73a82f237ff000d5053c8276d7df8683a79a4f625bf6f839366ca4a0519009a3","observation_id":"d5d67e8e-aa31-40b0-85f3-1718fcff4046","resolution":{"observed_at":"2026-08-06T14:45:20.607908Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:45:27.917140Z","title":"Plug-and-play: An efficient post-training pruningmethodforlargelanguagemodels","venue":null,"work_id":"4f3d616f-ae0d-47bf-b7fd-cae1680e2177","year":2024},"citing_paper":{"arxiv_id":"2507.18006","last_updated":"2025-07-24T00:49:48Z","snapshot_observed_at":"2026-08-07T13:43:37.969982Z","submitted_at":"2025-07-24T00:49:48Z","title":"Unlock the Potential of Fine-grained LLM Serving via Dynamic Module Scaling","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-06T14:45:20.612417Z"},"links":{"citing_paper":"/paper/2507.18006"},"observation_digest":"sha256:d45b965cdd36ab1752454c499960a85a50c6c140d2b0a106548d5a23de9868eb","observation_id":"8ed255ca-b436-478e-be84-cba993f9765f","resolution":{"observed_at":"2026-08-06T14:45:27.983499Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:45:27.740750Z","title":"Awq: Activation-aware weight quantization for llm compression and acceleration, 2024","venue":null,"work_id":"3b9ce18e-ed1e-46c2-a2f5-c014278813bc","year":2024},"citing_paper":{"arxiv_id":"2507.18006","last_updated":"2025-07-24T00:49:48Z","snapshot_observed_at":"2026-08-07T13:43:37.969982Z","submitted_at":"2025-07-24T00:49:48Z","title":"Unlock the Potential of Fine-grained LLM Serving via Dynamic Module Scaling","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-06T14:45:20.616704Z"},"links":{"citing_paper":"/paper/2507.18006"},"observation_digest":"sha256:d628f796daf43aa56c64292d73310fabd5dc01a0da4757f995ce8d3efe86fc40","observation_id":"226107fc-4b94-4798-8697-ebada942d07b","resolution":{"observed_at":"2026-08-06T14:45:27.805979Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:45:27.604722Z","title":"Zico Kolter","venue":null,"work_id":"c34f751d-b6a9-4365-b29e-dc8f107a7d46","year":2024},"citing_paper":{"arxiv_id":"2507.18006","last_updated":"2025-07-24T00:49:48Z","snapshot_observed_at":"2026-08-07T13:43:37.969982Z","submitted_at":"2025-07-24T00:49:48Z","title":"Unlock the Potential of Fine-grained LLM Serving via Dynamic Module Scaling","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-06T14:45:20.620795Z"},"links":{"citing_paper":"/paper/2507.18006"},"observation_digest":"sha256:fc5a7f42f001e568f66f8cdb45e90fd57e9c9b9294c3da82e9880d04d53ef5e4","observation_id":"bbac1862-4e2a-4f20-9b8f-9b4d6315ab69","resolution":{"observed_at":"2026-08-06T14:45:27.665893Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:45:27.425450Z","title":"Multiplexing dynamic deep learning workloads with slo-awareness in gpu clusters","venue":null,"work_id":"0092d6e8-4269-4aef-a46f-a8ab30ffbd9e","year":2025},"citing_paper":{"arxiv_id":"2507.18006","last_updated":"2025-07-24T00:49:48Z","snapshot_observed_at":"2026-08-07T13:43:37.969982Z","submitted_at":"2025-07-24T00:49:48Z","title":"Unlock the Potential of Fine-grained LLM Serving via Dynamic Module Scaling","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-06T14:45:20.624755Z"},"links":{"citing_paper":"/paper/2507.18006"},"observation_digest":"sha256:d36b02df60ca5f47d06d81a81d3d46db0f330c1d8bdc8a90dc83478d1b8b52f5","observation_id":"7f6173b9-8bbc-4803-a0ec-d755b9c17f1b","resolution":{"observed_at":"2026-08-06T14:45:27.515390Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:45:27.236012Z","title":"Cloudnativesim: A toolkit for modeling and simulation of cloud-native applications","venue":null,"work_id":"11371e9b-b7e6-4fee-984a-fa21e3993b79","year":2025},"citing_paper":{"arxiv_id":"2507.18006","last_updated":"2025-07-24T00:49:48Z","snapshot_observed_at":"2026-08-07T13:43:37.969982Z","submitted_at":"2025-07-24T00:49:48Z","title":"Unlock the Potential of Fine-grained LLM Serving via Dynamic Module Scaling","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-06T14:45:20.628509Z"},"links":{"citing_paper":"/paper/2507.18006"},"observation_digest":"sha256:b1e6a8b09ecbbc4927f8a43d47f0dfce4672a98803a990d3362b03a0340af0b4","observation_id":"43514fe2-a55b-48e3-862c-9f96811117c8","resolution":{"observed_at":"2026-08-06T14:45:27.346301Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:45:27.069834Z","title":"Llminaflash:Efficientlargelanguagemodelinferencewith limited memory, 2024","venue":null,"work_id":"87bc7391-35fa-4cd6-acac-42f4ba2e5f02","year":2024},"citing_paper":{"arxiv_id":"2507.18006","last_updated":"2025-07-24T00:49:48Z","snapshot_observed_at":"2026-08-07T13:43:37.969982Z","submitted_at":"2025-07-24T00:49:48Z","title":"Unlock the Potential of Fine-grained LLM Serving via Dynamic Module Scaling","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-06T14:45:20.632443Z"},"links":{"citing_paper":"/paper/2507.18006"},"observation_digest":"sha256:47723a45c7972cc79f3e4bb3f4c6e588684ee59b32d7b69eabcebb766fb1e5b5","observation_id":"ab4ad36b-d4ea-4784-8c2b-96f02774f583","resolution":{"observed_at":"2026-08-06T14:45:27.162975Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:45:20.636501Z","title":"Spotserve: Serving generative large language models on preemptible instances","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.18006","last_updated":"2025-07-24T00:49:48Z","snapshot_observed_at":"2026-08-07T13:43:37.969982Z","submitted_at":"2025-07-24T00:49:48Z","title":"Unlock the Potential of Fine-grained LLM Serving via Dynamic Module Scaling","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-06T14:45:20.636501Z"},"links":{"citing_paper":"/paper/2507.18006"},"observation_digest":"sha256:1fb830f4fc83198f1337e0e37fd3cda01adf6af2b038a18a8f1ebebff363422e","observation_id":"59bf1481-178e-4fc5-b5d8-ae5b96ba2e8a","resolution":{"observed_at":"2026-08-06T14:45:20.636501Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:45:26.890465Z","title":"Llumnix:Dynamicschedulingforlargelanguage model serving","venue":null,"work_id":"d6bedaf8-ca6c-4c1e-bd2d-fab78dcdf541","year":2024},"citing_paper":{"arxiv_id":"2507.18006","last_updated":"2025-07-24T00:49:48Z","snapshot_observed_at":"2026-08-07T13:43:37.969982Z","submitted_at":"2025-07-24T00:49:48Z","title":"Unlock the Potential of Fine-grained LLM Serving via Dynamic Module Scaling","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-06T14:45:20.640277Z"},"links":{"citing_paper":"/paper/2507.18006"},"observation_digest":"sha256:44a2a1037332f196fa719e5cd4827676d0387758a94454e47e7303ff7c928128","observation_id":"0736df2e-d30d-49e9-8f47-458134c404ed","resolution":{"observed_at":"2026-08-06T14:45:26.979547Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:45:26.701678Z","title":"Serving heterogeneous machine learning models on multi-gpu servers with spatio-temporal sharing","venue":null,"work_id":"9e53235e-9fa2-4d6c-adb9-967df0ac00f9","year":2022},"citing_paper":{"arxiv_id":"2507.18006","last_updated":"2025-07-24T00:49:48Z","snapshot_observed_at":"2026-08-07T13:43:37.969982Z","submitted_at":"2025-07-24T00:49:48Z","title":"Unlock the Potential of Fine-grained LLM Serving via Dynamic Module Scaling","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-06T14:45:20.643981Z"},"links":{"citing_paper":"/paper/2507.18006"},"observation_digest":"sha256:daa63c4a597e8d9250dc0a7006416017a6078981b8948699719a4438c486e041","observation_id":"559ffcaf-15c7-4c90-8d29-7712367d1853","resolution":{"observed_at":"2026-08-06T14:45:26.787021Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:45:26.529937Z","title":"Inferline:latency-aware provisioningandscalingforpredictionservingpipelines.In Proceedings of the 11th ACM Symposium on Cloud Computing, pages 477–491, 2020","venue":null,"work_id":"f16b045a-d77e-4250-ad8d-3b63ef95ce4c","year":2020},"citing_paper":{"arxiv_id":"2507.18006","last_updated":"2025-07-24T00:49:48Z","snapshot_observed_at":"2026-08-07T13:43:37.969982Z","submitted_at":"2025-07-24T00:49:48Z","title":"Unlock the Potential of Fine-grained LLM Serving via Dynamic Module Scaling","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-06T14:45:20.647609Z"},"links":{"citing_paper":"/paper/2507.18006"},"observation_digest":"sha256:87915ba13c163b5589d2ead12d9a65ab9fabe92c42056fae42d2ee29375e3fe0","observation_id":"9c86c373-b4bc-422f-b8be-c2e46e1f2c0b","resolution":{"observed_at":"2026-08-06T14:45:26.604327Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:45:26.350073Z","title":"Optimizing llm inference throughput via memory-aware and sla-constrained dynamic batching, 2025","venue":null,"work_id":"2f14bbc2-4922-41e2-aab3-dcc3c0f4fcfc","year":2025},"citing_paper":{"arxiv_id":"2507.18006","last_updated":"2025-07-24T00:49:48Z","snapshot_observed_at":"2026-08-07T13:43:37.969982Z","submitted_at":"2025-07-24T00:49:48Z","title":"Unlock the Potential of Fine-grained LLM Serving via Dynamic Module Scaling","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-06T14:45:20.651829Z"},"links":{"citing_paper":"/paper/2507.18006"},"observation_digest":"sha256:552d2123fe0067b2e433d68f879c5cec8018642de705b7ca23898ab98eadab11","observation_id":"c8937796-4580-4d32-a58c-fee8349597d3","resolution":{"observed_at":"2026-08-06T14:45:26.431926Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:45:26.152376Z","title":"Alloystack: A library operating system for serverless workflow applications.Pro- ceedings of the Twentieth European Conference on Computer Systems, 2025","venue":null,"work_id":"0f48f668-16e2-4402-9c36-ec26cca048c8","year":2025},"citing_paper":{"arxiv_id":"2507.18006","last_updated":"2025-07-24T00:49:48Z","snapshot_observed_at":"2026-08-07T13:43:37.969982Z","submitted_at":"2025-07-24T00:49:48Z","title":"Unlock the Potential of Fine-grained LLM Serving via Dynamic Module Scaling","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-06T14:45:20.655748Z"},"links":{"citing_paper":"/paper/2507.18006"},"observation_digest":"sha256:c7524d0b4358138526e897608b354e1fd0b918553f3be8b5a40a0d0dc2a5875b","observation_id":"ae8538a9-5dcf-4c02-88d2-141892950ff8","resolution":{"observed_at":"2026-08-06T14:45:26.270323Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:45:25.932844Z","title":"Lora-flow: Dynamic lora fusion for large lan- guage models in generative tasks","venue":null,"work_id":"b5743c40-b20c-4f95-95a4-15ff1dd753dc","year":2024},"citing_paper":{"arxiv_id":"2507.18006","last_updated":"2025-07-24T00:49:48Z","snapshot_observed_at":"2026-08-07T13:43:37.969982Z","submitted_at":"2025-07-24T00:49:48Z","title":"Unlock the Potential of Fine-grained LLM Serving via Dynamic Module Scaling","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-06T14:45:20.660020Z"},"links":{"citing_paper":"/paper/2507.18006"},"observation_digest":"sha256:ed0590d7e7da62d66b657010b8328845ef231f6106c3940745ce5af588fb6d06","observation_id":"0777050f-f079-44e5-8789-ffd80508e5ea","resolution":{"observed_at":"2026-08-06T14:45:26.052390Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:45:25.742506Z","title":"Accessed: Apr","venue":null,"work_id":"125b83f0-d9e3-4912-894d-4634a1b78efb","year":2025},"citing_paper":{"arxiv_id":"2507.18006","last_updated":"2025-07-24T00:49:48Z","snapshot_observed_at":"2026-08-07T13:43:37.969982Z","submitted_at":"2025-07-24T00:49:48Z","title":"Unlock the Potential of Fine-grained LLM Serving via Dynamic Module Scaling","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-06T14:45:20.663827Z"},"links":{"citing_paper":"/paper/2507.18006"},"observation_digest":"sha256:4b5f3d5791e273ff0dc8807257e6d65789987b7b6a66cd0fd731c3d033b254d2","observation_id":"f1344fe9-90df-4ba6-8d48-ab40f74d133f","resolution":{"observed_at":"2026-08-06T14:45:25.830080Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:45:25.567172Z","title":null,"venue":null,"work_id":"3f8fd8fe-189a-4a6b-8862-653fce8b6bfb","year":2020},"citing_paper":{"arxiv_id":"2507.18006","last_updated":"2025-07-24T00:49:48Z","snapshot_observed_at":"2026-08-07T13:43:37.969982Z","submitted_at":"2025-07-24T00:49:48Z","title":"Unlock the Potential of Fine-grained LLM Serving via Dynamic Module Scaling","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-06T14:45:20.667723Z"},"links":{"citing_paper":"/paper/2507.18006"},"observation_digest":"sha256:226c4b0178cfb65abd1f91da9c5f35fb591007f8941112117af9ae40551046f1","observation_id":"01a69b05-5d9b-463a-b295-4217399f77da","resolution":{"observed_at":"2026-08-06T14:45:25.642539Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:45:25.404456Z","title":null,"venue":null,"work_id":"99ef0896-5a21-4f70-9276-3c30d6709540","year":2026},"citing_paper":{"arxiv_id":"2507.18006","last_updated":"2025-07-24T00:49:48Z","snapshot_observed_at":"2026-08-07T13:43:37.969982Z","submitted_at":"2025-07-24T00:49:48Z","title":"Unlock the Potential of Fine-grained LLM Serving via Dynamic Module Scaling","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-06T14:45:20.675885Z"},"links":{"citing_paper":"/paper/2507.18006"},"observation_digest":"sha256:2e6b468ccf5b9ac6d0b2b9f2ac5b898d198aeea13a1474d311e42913706277a3","observation_id":"2213fac0-1631-4eec-bdc7-3978d02e6221","resolution":{"observed_at":"2026-08-06T14:45:25.486001Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:45:20.681562Z","title":"Attention is all you need.Advances in neural information processing systems, 30, 2017","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2507.18006","last_updated":"2025-07-24T00:49:48Z","snapshot_observed_at":"2026-08-07T13:43:37.969982Z","submitted_at":"2025-07-24T00:49:48Z","title":"Unlock the Potential of Fine-grained LLM Serving via Dynamic Module Scaling","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-06T14:45:20.681562Z"},"links":{"citing_paper":"/paper/2507.18006"},"observation_digest":"sha256:ea5ad86aba0c6fc7b78cebb371cd32c5ecab3d15b311241cc5ecd306f6fa4fe0","observation_id":"1421a709-5b0d-4168-99c6-cff475463e4b","resolution":{"observed_at":"2026-08-06T14:45:20.681562Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:45:25.200271Z","title":"Llm inference serving: Survey of recent advances and opportunities, 2024","venue":null,"work_id":"9d24d35c-5822-429b-95ac-572e8df8f086","year":2024},"citing_paper":{"arxiv_id":"2507.18006","last_updated":"2025-07-24T00:49:48Z","snapshot_observed_at":"2026-08-07T13:43:37.969982Z","submitted_at":"2025-07-24T00:49:48Z","title":"Unlock the Potential of Fine-grained LLM Serving via Dynamic Module Scaling","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-06T14:45:20.686206Z"},"links":{"citing_paper":"/paper/2507.18006"},"observation_digest":"sha256:5bc30dc80a35e2539174196e6eb746e3c1653f13d511479e09bfa87e98b0a386","observation_id":"25fa635e-8597-4a2c-a108-282095a2f577","resolution":{"observed_at":"2026-08-06T14:45:25.265670Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:45:25.015072Z","title":"Optimizing mixture-of-experts inference time combining model deployment and communication scheduling, 2024","venue":null,"work_id":"f224f3ff-dee9-4fbe-b878-50bc30c65755","year":2024},"citing_paper":{"arxiv_id":"2507.18006","last_updated":"2025-07-24T00:49:48Z","snapshot_observed_at":"2026-08-07T13:43:37.969982Z","submitted_at":"2025-07-24T00:49:48Z","title":"Unlock the Potential of Fine-grained LLM Serving via Dynamic Module Scaling","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-06T14:45:20.690334Z"},"links":{"citing_paper":"/paper/2507.18006"},"observation_digest":"sha256:d553041fee507c7600ec1d13d5578c6f637b0838b0f7deaba04c75778bb9656e","observation_id":"85001b70-bf8c-41bd-9932-9291d297961a","resolution":{"observed_at":"2026-08-06T14:45:25.101053Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:45:24.817562Z","title":"Mooncake: A kvcache-centric disag- gregated architecture for llm serving, 2024","venue":null,"work_id":"1448b175-25d0-4359-8fa8-54f5a296e421","year":2024},"citing_paper":{"arxiv_id":"2507.18006","last_updated":"2025-07-24T00:49:48Z","snapshot_observed_at":"2026-08-07T13:43:37.969982Z","submitted_at":"2025-07-24T00:49:48Z","title":"Unlock the Potential of Fine-grained LLM Serving via Dynamic Module Scaling","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-06T14:45:20.694330Z"},"links":{"citing_paper":"/paper/2507.18006"},"observation_digest":"sha256:932daae27fd7d1a69b7bb150d35bda0d3f1f0a403bd428d72e399195da86d879","observation_id":"3df35fa0-3e6a-4d50-95ce-866af36fc334","resolution":{"observed_at":"2026-08-06T14:45:24.909654Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:45:24.593539Z","title":"Spinfer: Leveraging low-level sparsity for efficient large language model inference on gpus","venue":null,"work_id":"b6c4de4c-8ef8-41e4-9796-28a287f9e3ed","year":2025},"citing_paper":{"arxiv_id":"2507.18006","last_updated":"2025-07-24T00:49:48Z","snapshot_observed_at":"2026-08-07T13:43:37.969982Z","submitted_at":"2025-07-24T00:49:48Z","title":"Unlock the Potential of Fine-grained LLM Serving via Dynamic Module Scaling","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-06T14:45:20.698281Z"},"links":{"citing_paper":"/paper/2507.18006"},"observation_digest":"sha256:e24827720e55bc72f07f43b90371570d9859a847305d88877317fdf72c2d6fbb","observation_id":"ec14dcb2-cd43-4691-8eb5-08930a6dfee8","resolution":{"observed_at":"2026-08-06T14:45:24.699198Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:45:24.443357Z","title":"H2o:Heavy-hitteroracleforefficientgenerativeinference of large language models.Advances in Neural Information Processing Systems, 36:34661–34710, 2023","venue":null,"work_id":"d3bea8ba-ff2e-4623-9ecc-682b30e48b7e","year":2023},"citing_paper":{"arxiv_id":"2507.18006","last_updated":"2025-07-24T00:49:48Z","snapshot_observed_at":"2026-08-07T13:43:37.969982Z","submitted_at":"2025-07-24T00:49:48Z","title":"Unlock the Potential of Fine-grained LLM Serving via Dynamic Module Scaling","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-06T14:45:20.702463Z"},"links":{"citing_paper":"/paper/2507.18006"},"observation_digest":"sha256:ededc3eecab767bd7183fc4d74805446d0f092824343dadb7ab53393d3ac4e08","observation_id":"1ecc3806-31e9-491d-ae80-e0550faa2b0b","resolution":{"observed_at":"2026-08-06T14:45:24.511967Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:45:24.188936Z","title":"Splitwise: Efficient gen- erative llm inference using phase splitting.2024 ACM/IEEE 51st Annual International Symposium on Computer Architecture (ISCA), pages 118–132, 2023","venue":null,"work_id":"82fd02ad-a25a-4b07-ac7d-b00208d14148","year":2024},"citing_paper":{"arxiv_id":"2507.18006","last_updated":"2025-07-24T00:49:48Z","snapshot_observed_at":"2026-08-07T13:43:37.969982Z","submitted_at":"2025-07-24T00:49:48Z","title":"Unlock the Potential of Fine-grained LLM Serving via Dynamic Module Scaling","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-06T14:45:20.706607Z"},"links":{"citing_paper":"/paper/2507.18006"},"observation_digest":"sha256:6db2b6bafabe2aa1a49e09c728c30c191ed8ce9837a29d85d8ae12804d25f27e","observation_id":"76faaf30-a734-4153-8442-4a54223cfa77","resolution":{"observed_at":"2026-08-06T14:45:24.286745Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:45:23.993448Z","title":null,"venue":null,"work_id":"e40c27fe-6168-4d92-aa53-2bb49dd57db8","year":2024},"citing_paper":{"arxiv_id":"2507.18006","last_updated":"2025-07-24T00:49:48Z","snapshot_observed_at":"2026-08-07T13:43:37.969982Z","submitted_at":"2025-07-24T00:49:48Z","title":"Unlock the Potential of Fine-grained LLM Serving via Dynamic Module Scaling","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-06T14:45:20.710774Z"},"links":{"citing_paper":"/paper/2507.18006"},"observation_digest":"sha256:c4fcca3d12888a1cce3cd9c614eb70f446ad9f159c6c55f33bdeac9889d17815","observation_id":"c4c725c7-5a59-4d32-8bf4-3e95d7168ab7","resolution":{"observed_at":"2026-08-06T14:45:24.086241Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:45:23.751104Z","title":"Inference without interference: Disaggregate llm inference for mixed downstream workloads, 2024","venue":null,"work_id":"c1cdd3dd-8988-4066-92be-42f0fb76398f","year":2024},"citing_paper":{"arxiv_id":"2507.18006","last_updated":"2025-07-24T00:49:48Z","snapshot_observed_at":"2026-08-07T13:43:37.969982Z","submitted_at":"2025-07-24T00:49:48Z","title":"Unlock the Potential of Fine-grained LLM Serving via Dynamic Module Scaling","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-06T14:45:20.714736Z"},"links":{"citing_paper":"/paper/2507.18006"},"observation_digest":"sha256:dbcb2f39a3323f1dc2f3c9a923672b65826fdeb188926d34c09420dd570f3262","observation_id":"43ce76af-0cca-4fae-bae4-1514640d0afa","resolution":{"observed_at":"2026-08-06T14:45:23.843614Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:45:23.542572Z","title":"Dynamollm:Designingllminferenceclustersforperformance and energy efficiency, 2024","venue":null,"work_id":"5ec41813-a7b0-4652-9b6f-59d788f941e7","year":2024},"citing_paper":{"arxiv_id":"2507.18006","last_updated":"2025-07-24T00:49:48Z","snapshot_observed_at":"2026-08-07T13:43:37.969982Z","submitted_at":"2025-07-24T00:49:48Z","title":"Unlock the Potential of Fine-grained LLM Serving via Dynamic Module Scaling","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-06T14:45:20.719014Z"},"links":{"citing_paper":"/paper/2507.18006"},"observation_digest":"sha256:1b5f465caf72aebc1788aa87164937497c5b573ddf9a0c1524739e5282d0d709","observation_id":"904791a1-6169-4a39-a9f4-18cb5f46ec30","resolution":{"observed_at":"2026-08-06T14:45:23.669168Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:45:23.333940Z","title":"Skyserve:Servingaimodels across regions and clouds with spot instances","venue":null,"work_id":"02d67659-8402-4f50-a328-d9862b7de7f7","year":2025},"citing_paper":{"arxiv_id":"2507.18006","last_updated":"2025-07-24T00:49:48Z","snapshot_observed_at":"2026-08-07T13:43:37.969982Z","submitted_at":"2025-07-24T00:49:48Z","title":"Unlock the Potential of Fine-grained LLM Serving via Dynamic Module Scaling","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-06T14:45:20.722857Z"},"links":{"citing_paper":"/paper/2507.18006"},"observation_digest":"sha256:b21670305ae03a39658b5ae676b45d10ad3acf869c976b34cc2e916cd306aa3f","observation_id":"d9ebb7f2-f091-404d-8285-19b0415de7b8","resolution":{"observed_at":"2026-08-06T14:45:23.437689Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:45:23.119873Z","title":"Towardsefficientandreliablellmserving: A real-world workload study.arXiv preprint arXiv:2402.XXXXX, 2024","venue":null,"work_id":"1ba45b8f-686a-46c8-990d-4695d481b355","year":2024},"citing_paper":{"arxiv_id":"2507.18006","last_updated":"2025-07-24T00:49:48Z","snapshot_observed_at":"2026-08-07T13:43:37.969982Z","submitted_at":"2025-07-24T00:49:48Z","title":"Unlock the Potential of Fine-grained LLM Serving via Dynamic Module Scaling","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-06T14:45:20.727165Z"},"links":{"citing_paper":"/paper/2507.18006"},"observation_digest":"sha256:b274f3439b8d50dd0923adcd444a4c7c9eb381b8f419358a67635b27203cce91","observation_id":"d650dde7-70e7-4800-9354-556790f414ee","resolution":{"observed_at":"2026-08-06T14:45:23.227733Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:45:22.917236Z","title":"Usher: Holistic interference avoidance for resource optimized ML inference","venue":null,"work_id":"2b2a710e-d63d-4e47-aeeb-d13acae71e55","year":2024},"citing_paper":{"arxiv_id":"2507.18006","last_updated":"2025-07-24T00:49:48Z","snapshot_observed_at":"2026-08-07T13:43:37.969982Z","submitted_at":"2025-07-24T00:49:48Z","title":"Unlock the Potential of Fine-grained LLM Serving via Dynamic Module Scaling","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-06T14:45:20.731400Z"},"links":{"citing_paper":"/paper/2507.18006"},"observation_digest":"sha256:a2c25e7488b41209e09f496d810f39a74121bba57c23b0c1edcaf2193c35b659","observation_id":"50e2209c-e0bc-4164-972b-9d186e868a2e","resolution":{"observed_at":"2026-08-06T14:45:23.012903Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:45:22.712257Z","title":null,"venue":null,"work_id":"3a96d361-f111-4111-9521-b881ebfcb3cd","year":2024},"citing_paper":{"arxiv_id":"2507.18006","last_updated":"2025-07-24T00:49:48Z","snapshot_observed_at":"2026-08-07T13:43:37.969982Z","submitted_at":"2025-07-24T00:49:48Z","title":"Unlock the Potential of Fine-grained LLM Serving via Dynamic Module Scaling","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-06T14:45:20.735534Z"},"links":{"citing_paper":"/paper/2507.18006"},"observation_digest":"sha256:d27a4fead964bd73ebe5b2a479178c59b1948ca485e264c98d95693162f1ae7e","observation_id":"71fe8728-f510-4dde-8c3a-93b4cea88868","resolution":{"observed_at":"2026-08-06T14:45:22.779244Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:45:22.472864Z","title":"Distserve: Disaggregating prefill and decoding for goodput-optimized large language model serving","venue":null,"work_id":"ff9513b3-0ee1-4649-9c86-eb3e44db5cd2","year":2024},"citing_paper":{"arxiv_id":"2507.18006","last_updated":"2025-07-24T00:49:48Z","snapshot_observed_at":"2026-08-07T13:43:37.969982Z","submitted_at":"2025-07-24T00:49:48Z","title":"Unlock the Potential of Fine-grained LLM Serving via Dynamic Module Scaling","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-06T14:45:20.739514Z"},"links":{"citing_paper":"/paper/2507.18006"},"observation_digest":"sha256:edb2791d916b399ede08563de5e3ae5864f43d62a4daeb133f11e52dde73766d","observation_id":"515dbeab-5865-4a1e-89d1-4b2dbc2a8595","resolution":{"observed_at":"2026-08-06T14:45:22.599134Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:45:22.314614Z","title":"Alpaserve: Statistical multiplexing with model parallelism for deep learning serving","venue":null,"work_id":"d1f61fe4-abef-49ed-88fe-fcb2a5fa6ca7","year":2023},"citing_paper":{"arxiv_id":"2507.18006","last_updated":"2025-07-24T00:49:48Z","snapshot_observed_at":"2026-08-07T13:43:37.969982Z","submitted_at":"2025-07-24T00:49:48Z","title":"Unlock the Potential of Fine-grained LLM Serving via Dynamic Module Scaling","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-06T14:45:20.743488Z"},"links":{"citing_paper":"/paper/2507.18006"},"observation_digest":"sha256:407a9dca8d70644f9a56be1bf55cf8e7fc0958824d6048a0e05fc5850a779e94","observation_id":"b17279ee-733a-4137-87dc-7c2105163f89","resolution":{"observed_at":"2026-08-06T14:45:22.387183Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:45:22.108310Z","title":null,"venue":null,"work_id":"fd5106b5-2045-4cc8-8d31-b2728d087b2b","year":2020},"citing_paper":{"arxiv_id":"2507.18006","last_updated":"2025-07-24T00:49:48Z","snapshot_observed_at":"2026-08-07T13:43:37.969982Z","submitted_at":"2025-07-24T00:49:48Z","title":"Unlock the Potential of Fine-grained LLM Serving via Dynamic Module Scaling","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-06T14:45:20.747191Z"},"links":{"citing_paper":"/paper/2507.18006"},"observation_digest":"sha256:b96227bc7891d3d53bf5438f0953676cabd7a9335cc391abf62299c5fc0c47b2","observation_id":"3db551ec-4c4a-449c-8a13-15ea022f6fdf","resolution":{"observed_at":"2026-08-06T14:45:22.205088Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:45:21.935781Z","title":"Yadwadkar, and Christos Kozyrakis","venue":null,"work_id":"9e276474-2fe1-4202-a25f-7be9b4f05658","year":null},"citing_paper":{"arxiv_id":"2507.18006","last_updated":"2025-07-24T00:49:48Z","snapshot_observed_at":"2026-08-07T13:43:37.969982Z","submitted_at":"2025-07-24T00:49:48Z","title":"Unlock the Potential of Fine-grained LLM Serving via Dynamic Module Scaling","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-06T14:45:20.750754Z"},"links":{"citing_paper":"/paper/2507.18006"},"observation_digest":"sha256:dae713d38c2d5f38a44977a34e666d001c730c21210e09c138374dc173984800","observation_id":"8e952f21-82e8-4dfb-aeab-26bb3c1c3e81","resolution":{"observed_at":"2026-08-06T14:45:21.999977Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2205.05198","last_updated":"2022-05-10T22:40:17Z","snapshot_observed_at":"2026-07-06T13:08:41.837840Z","submitted_at":"2022-05-10T22:40:17Z","title":"Reducing Activation Recomputation in Large Transformer Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2205.05198","snapshot_observed_at":"2026-08-06T14:45:20.758677Z","title":"McAfee,MichaelAndersch,MohammadShoeybi,andBryanCatanzaro","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2507.18006","last_updated":"2025-07-24T00:49:48Z","snapshot_observed_at":"2026-08-07T13:43:37.969982Z","submitted_at":"2025-07-24T00:49:48Z","title":"Unlock the Potential of Fine-grained LLM Serving via Dynamic Module Scaling","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-06T14:45:20.758677Z"},"links":{"cited_paper":"/paper/2205.05198","citing_paper":"/paper/2507.18006"},"observation_digest":"sha256:654bd22e5bf2917cd6eac416829ffb382b50ee83353f0b70dfb43a3308b7aa04","observation_id":"29feca7a-4ba7-4d1c-870a-f553e4fdb36f","resolution":{"observed_at":"2026-08-06T14:45:20.758677Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:45:21.535471Z","title":"On parallel processing systems: Amdahl’s law generalized and some results on optimal design","venue":null,"work_id":"0d751978-721a-42d5-a4b5-f9bc21adf0fa","year":1992},"citing_paper":{"arxiv_id":"2507.18006","last_updated":"2025-07-24T00:49:48Z","snapshot_observed_at":"2026-08-07T13:43:37.969982Z","submitted_at":"2025-07-24T00:49:48Z","title":"Unlock the Potential of Fine-grained LLM Serving via Dynamic Module Scaling","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-06T14:45:20.762745Z"},"links":{"citing_paper":"/paper/2507.18006"},"observation_digest":"sha256:970eb92826275ca88a7e823d03987f47e94bf353a4826fe3a9c29cd7a8afde67","observation_id":"6192e0ab-e576-407f-938f-4a21570f9f37","resolution":{"observed_at":"2026-08-06T14:45:21.610721Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:45:21.330409Z","title":"xformers: A modular and hackable transformer modelling library.https://github.com/facebookresearch/ xformers, 2022","venue":null,"work_id":"dabfc945-d31b-45a6-9169-5553adf6e5d9","year":2022},"citing_paper":{"arxiv_id":"2507.18006","last_updated":"2025-07-24T00:49:48Z","snapshot_observed_at":"2026-08-07T13:43:37.969982Z","submitted_at":"2025-07-24T00:49:48Z","title":"Unlock the Potential of Fine-grained LLM Serving via Dynamic Module Scaling","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-06T14:45:20.766581Z"},"links":{"citing_paper":"/paper/2507.18006"},"observation_digest":"sha256:3640e53988041622177dedd306a15b6b14c76e77cd7d9659c9ce98aa882d8142","observation_id":"3607326f-b68b-41b0-bfc0-29a9cb5085ec","resolution":{"observed_at":"2026-08-06T14:45:21.412262Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:45:21.305012Z","title":"https://developer.nvidia.com/management-library-nvml, 2025","venue":null,"work_id":"4e755e41-d952-4ebf-916c-8f76315acea8","year":2025},"citing_paper":{"arxiv_id":"2507.18006","last_updated":"2025-07-24T00:49:48Z","snapshot_observed_at":"2026-08-07T13:43:37.969982Z","submitted_at":"2025-07-24T00:49:48Z","title":"Unlock the Potential of Fine-grained LLM Serving via Dynamic Module Scaling","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-06T14:45:20.770284Z"},"links":{"citing_paper":"/paper/2507.18006"},"observation_digest":"sha256:3eccfc692860b14495e6aaa70821285e147bb3eef90a7e6f3430e46e9f6d27af","observation_id":"b995e9ff-346e-43c5-a422-8101a97dc13e","resolution":{"observed_at":"2026-08-06T14:45:21.313583Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:45:21.272261Z","title":"Orca: A distributed serving system for transformer- based generative models","venue":null,"work_id":"55ac6ba0-20e8-4e53-ba66-36957a4d1409","year":2022},"citing_paper":{"arxiv_id":"2507.18006","last_updated":"2025-07-24T00:49:48Z","snapshot_observed_at":"2026-08-07T13:43:37.969982Z","submitted_at":"2025-07-24T00:49:48Z","title":"Unlock the Potential of Fine-grained LLM Serving via Dynamic Module Scaling","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-06T14:45:20.774358Z"},"links":{"citing_paper":"/paper/2507.18006"},"observation_digest":"sha256:8e4cb8caac24d189371724f50a2c7963be7aa7e588a1740958c323d5dfa2aceb","observation_id":"33dca7d2-896f-4afa-a787-f6a83496e92f","resolution":{"observed_at":"2026-08-06T14:45:21.283738Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:45:21.231279Z","title":"Uellm: A unified and efficient approach for large language model inference serving","venue":null,"work_id":"720a43b6-dea3-4919-a8e2-0316374c801b","year":2024},"citing_paper":{"arxiv_id":"2507.18006","last_updated":"2025-07-24T00:49:48Z","snapshot_observed_at":"2026-08-07T13:43:37.969982Z","submitted_at":"2025-07-24T00:49:48Z","title":"Unlock the Potential of Fine-grained LLM Serving via Dynamic Module Scaling","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-06T14:45:20.778185Z"},"links":{"citing_paper":"/paper/2507.18006"},"observation_digest":"sha256:ef0a4d7d9a4282dd9690c9785ce0a34791dcd155689c8a0f58e1a19f991efcbe","observation_id":"080e1029-b4cb-43b7-9733-66fb8c5041d6","resolution":{"observed_at":"2026-08-06T14:45:21.245867Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:45:21.185558Z","title":"Stanford alpaca: An instruction- following llama model.https://github.com/tatsu-lab/stanford_alpaca,","venue":null,"work_id":"24e2a8eb-a32e-43b6-823f-08a9037cb252","year":null},"citing_paper":{"arxiv_id":"2507.18006","last_updated":"2025-07-24T00:49:48Z","snapshot_observed_at":"2026-08-07T13:43:37.969982Z","submitted_at":"2025-07-24T00:49:48Z","title":"Unlock the Potential of Fine-grained LLM Serving via Dynamic Module Scaling","version":1},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-06T14:45:20.781963Z"},"links":{"citing_paper":"/paper/2507.18006"},"observation_digest":"sha256:fa8cd6bda7f96a6a3b0b490e22dfda8f58639ce1bba01e5a5e659efb8f9cc976","observation_id":"85ae5a38-9a43-400b-9e93-4dcf3748b8cf","resolution":{"observed_at":"2026-08-06T14:45:21.202598Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:45:21.096825Z","title":"Mepipe: Democratizing llm training with memory- efficient slice-level pipeline scheduling on cost-effective accelerators","venue":null,"work_id":"9df4ad4a-b9db-46dc-8b95-03a52ff6145e","year":2025},"citing_paper":{"arxiv_id":"2507.18006","last_updated":"2025-07-24T00:49:48Z","snapshot_observed_at":"2026-08-07T13:43:37.969982Z","submitted_at":"2025-07-24T00:49:48Z","title":"Unlock the Potential of Fine-grained LLM Serving via Dynamic Module Scaling","version":1},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-06T14:45:20.789659Z"},"links":{"citing_paper":"/paper/2507.18006"},"observation_digest":"sha256:151d00d7c766f3d35d65ba50de568fffb353ed55891bb21ab0a0f1bdd68057ff","observation_id":"f8ce4997-8690-4e79-92bb-6d436cbc5305","resolution":{"observed_at":"2026-08-06T14:45:21.113047Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:45:21.065432Z","title":"Le, and Z","venue":null,"work_id":"6b1b471b-39b0-444d-ad54-f032e29adca6","year":2018},"citing_paper":{"arxiv_id":"2507.18006","last_updated":"2025-07-24T00:49:48Z","snapshot_observed_at":"2026-08-07T13:43:37.969982Z","submitted_at":"2025-07-24T00:49:48Z","title":"Unlock the Potential of Fine-grained LLM Serving via Dynamic Module Scaling","version":1},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-08-06T14:45:20.793446Z"},"links":{"citing_paper":"/paper/2507.18006"},"observation_digest":"sha256:cdf4140a5ff788d7b4653f7b022d3a6010fb09777dd17dd7c2ef9e248f720656","observation_id":"d44c4132-fcb8-4e9b-b5eb-fe849f84db51","resolution":{"observed_at":"2026-08-06T14:45:21.078885Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:45:21.051154Z","title":"Mist:Efficientdistributedtrainingoflargelanguagemodelsviamemory- parallelism co-optimization","venue":null,"work_id":"21746a6f-4254-4c80-9ec1-6b425d5e67a6","year":2025},"citing_paper":{"arxiv_id":"2507.18006","last_updated":"2025-07-24T00:49:48Z","snapshot_observed_at":"2026-08-07T13:43:37.969982Z","submitted_at":"2025-07-24T00:49:48Z","title":"Unlock the Potential of Fine-grained LLM Serving via Dynamic Module Scaling","version":1},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-06T14:45:20.797367Z"},"links":{"citing_paper":"/paper/2507.18006"},"observation_digest":"sha256:18258bd47fb1dc04c71be71b15693f8a4955e29c0fd2c18a71ca9a73a743934a","observation_id":"ab48421c-b07c-4781-905c-e160553b1ac6","resolution":{"observed_at":"2026-08-06T14:45:21.055530Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1909.08053","last_updated":"2020-03-13T23:45:18Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2019-09-17T19:42:54Z","title":"Megatron-LM: Training Multi-Billion Parameter Language Models Using Model Parallelism","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1909.08053","snapshot_observed_at":"2026-08-06T14:45:20.801131Z","title":"14 Unlock the Potential of Fine-grained LLM Serving via Dynamic Module Scaling EuroSys ’26, April 13–April 16, 2026, Edinburgh, UK","venue":null,"work_id":null,"year":1909},"citing_paper":{"arxiv_id":"2507.18006","last_updated":"2025-07-24T00:49:48Z","snapshot_observed_at":"2026-08-07T13:43:37.969982Z","submitted_at":"2025-07-24T00:49:48Z","title":"Unlock the Potential of Fine-grained LLM Serving via Dynamic Module Scaling","version":1},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-08-06T14:45:20.801131Z"},"links":{"cited_paper":"/paper/1909.08053","citing_paper":"/paper/2507.18006"},"observation_digest":"sha256:c92e1cd7244c913d30085855a7abbc36313b551f46e94ef7e50ba2ded7244cc1","observation_id":"b2b93895-5df9-4f87-91b4-4b5a78aab060","resolution":{"observed_at":"2026-08-06T14:45:20.801131Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:45:21.037879Z","title":"Alpa: Automating inter and intra- operator parallelism for distributed deep learning","venue":null,"work_id":"e7c17256-3a88-4668-85cc-69e39493f87f","year":2022},"citing_paper":{"arxiv_id":"2507.18006","last_updated":"2025-07-24T00:49:48Z","snapshot_observed_at":"2026-08-07T13:43:37.969982Z","submitted_at":"2025-07-24T00:49:48Z","title":"Unlock the Potential of Fine-grained LLM Serving via Dynamic Module Scaling","version":1},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-08-06T14:45:20.805316Z"},"links":{"citing_paper":"/paper/2507.18006"},"observation_digest":"sha256:4c36264fb077a0ebf3094724a0c57c643831781e78614c9c9efd76f07e0e1d1d","observation_id":"848a515f-29f3-418e-8124-111d46d06335","resolution":{"observed_at":"2026-08-06T14:45:21.041953Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:45:21.023917Z","title":"Fast state restoration in llm serving with hcache","venue":null,"work_id":"8cfe9f7b-43d8-43c9-bb63-7ce5de673b0b","year":2025},"citing_paper":{"arxiv_id":"2507.18006","last_updated":"2025-07-24T00:49:48Z","snapshot_observed_at":"2026-08-07T13:43:37.969982Z","submitted_at":"2025-07-24T00:49:48Z","title":"Unlock the Potential of Fine-grained LLM Serving via Dynamic Module Scaling","version":1},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-08-06T14:45:20.809631Z"},"links":{"citing_paper":"/paper/2507.18006"},"observation_digest":"sha256:f9738b1c1e46713cd2d18c97c9cf083c159dbf21416a540c937deccc09d79fb5","observation_id":"a7567b5a-d5ea-4aff-aee2-d1c5c27566fc","resolution":{"observed_at":"2026-08-06T14:45:21.028193Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:45:21.009772Z","title":"Fast and live model auto scaling with o(1) host caching, 2024","venue":null,"work_id":"a3b3460f-f2a5-4d0f-874c-c85138451556","year":2024},"citing_paper":{"arxiv_id":"2507.18006","last_updated":"2025-07-24T00:49:48Z","snapshot_observed_at":"2026-08-07T13:43:37.969982Z","submitted_at":"2025-07-24T00:49:48Z","title":"Unlock the Potential of Fine-grained LLM Serving via Dynamic Module Scaling","version":1},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-08-06T14:45:20.813723Z"},"links":{"citing_paper":"/paper/2507.18006"},"observation_digest":"sha256:a7edf3e1ed0bb3a8460909010d2cc65071938ad33efec17ddc3d3b1bcea4b29e","observation_id":"5b2ca156-3b79-4afc-8b43-a3107e2d25aa","resolution":{"observed_at":"2026-08-06T14:45:21.013948Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:45:20.817827Z","title":"Deepspeed: System optimizations enable training deep learning models with over 100 billion parameters","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2507.18006","last_updated":"2025-07-24T00:49:48Z","snapshot_observed_at":"2026-08-07T13:43:37.969982Z","submitted_at":"2025-07-24T00:49:48Z","title":"Unlock the Potential of Fine-grained LLM Serving via Dynamic Module Scaling","version":1},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-08-06T14:45:20.817827Z"},"links":{"citing_paper":"/paper/2507.18006"},"observation_digest":"sha256:d6924bb170054686d52cda02c076c3260415198d8c16fc7fca54295378d07955","observation_id":"106593b2-91f9-4854-87af-93f5faf2cc46","resolution":{"observed_at":"2026-08-06T14:45:20.817827Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:45:20.985086Z","title":"Infinigen: Efficientgenerativeinferenceoflargelanguagemodelswithdynamickv cache management","venue":null,"work_id":"68e06269-d815-474a-bc3a-cb0390e5bfad","year":2024},"citing_paper":{"arxiv_id":"2507.18006","last_updated":"2025-07-24T00:49:48Z","snapshot_observed_at":"2026-08-07T13:43:37.969982Z","submitted_at":"2025-07-24T00:49:48Z","title":"Unlock the Potential of Fine-grained LLM Serving via Dynamic Module Scaling","version":1},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-08-06T14:45:20.821661Z"},"links":{"citing_paper":"/paper/2507.18006"},"observation_digest":"sha256:7a4d0f990d4b739feb7a7b90969000c076936f6c2121b2ea95ffc594e09c56e6","observation_id":"d07250cf-5f05-4c41-8fdb-e4ecfcde3f15","resolution":{"observed_at":"2026-08-06T14:45:20.991136Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:45:21.722126Z","title":null,"venue":null,"work_id":"3eb55733-6552-4e63-93f8-414c416079d0","year":2021},"citing_paper":{"arxiv_id":"2507.18006","last_updated":"2025-07-24T00:49:48Z","snapshot_observed_at":"2026-08-07T13:43:37.969982Z","submitted_at":"2025-07-24T00:49:48Z","title":"Unlock the Potential of Fine-grained LLM Serving via Dynamic Module Scaling","version":1},"reference_index":411,"source":"pdf_text","source_observed_at":"2026-08-06T14:45:20.754810Z"},"links":{"citing_paper":"/paper/2507.18006"},"observation_digest":"sha256:78d0e060c2e70248a60716ac58a0a0e2b7888ab4c16688a1ab4e51030ef000e3","observation_id":"bb71e73a-2187-431a-8428-e8c3931b8909","resolution":{"observed_at":"2026-08-06T14:45:21.824444Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:45:20.671778Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.18006","last_updated":"2025-07-24T00:49:48Z","snapshot_observed_at":"2026-08-07T13:43:37.969982Z","submitted_at":"2025-07-24T00:49:48Z","title":"Unlock the Potential of Fine-grained LLM Serving via Dynamic Module Scaling","version":1},"reference_index":2020,"source":"pdf_text","source_observed_at":"2026-08-06T14:45:20.671778Z"},"links":{"citing_paper":"/paper/2507.18006"},"observation_digest":"sha256:85c33241e8d9e9daf362cf195f5351daa95293eedd0c83513753ebe9a2032e6a","observation_id":"a18aea7a-6abb-454a-a786-988f05df980b","resolution":{"observed_at":"2026-08-06T14:45:20.671778Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T14:45:21.127827Z","title":null,"venue":null,"work_id":"10b6ee2c-e93b-4b54-aa96-5e824aff3f72","year":2025},"citing_paper":{"arxiv_id":"2507.18006","last_updated":"2025-07-24T00:49:48Z","snapshot_observed_at":"2026-08-07T13:43:37.969982Z","submitted_at":"2025-07-24T00:49:48Z","title":"Unlock the Potential of Fine-grained LLM Serving via Dynamic Module Scaling","version":1},"reference_index":2023,"source":"pdf_text","source_observed_at":"2026-08-06T14:45:20.785956Z"},"links":{"citing_paper":"/paper/2507.18006"},"observation_digest":"sha256:bdd945d55ef47bd8de69419a9bfea245460f9c65a0d5d7a1ea6ae14402932291","observation_id":"2ef497c9-742a-44aa-9990-dbfd9f1c122c","resolution":{"observed_at":"2026-08-06T14:45:21.165178Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2507.18006","last_updated":"2025-07-24T00:49:48Z","latest_version":1,"primary_category":"cs.DC","snapshot_observed_at":"2026-08-07T13:43:37.969982Z","submitted_at":"2025-07-24T00:49:48Z","title":"Unlock the Potential of Fine-grained LLM Serving via Dynamic Module Scaling"},"reference_resolution":{"displayed":64,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":21,"verified_exact":0,"verified_fuzzy":43},"total_outbound_references":64},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"thesis":"As of 7 August 2026, this Paper Citation Record lists 64 of 64 outbound references and 0 inbound Pith citation observations for arXiv:2507.18006."}