{"as_of":"2026-08-07T16:41:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:d4607097fe01b52c8798caf54d3290dced0f888609b1f053f7b7347e9265bbb6","coverage":[{"denominator":39,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":39,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-03T04:22:54.961186Z","state":"measured"},{"denominator":39,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":39,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-07T06:34:17.273281+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2607.29575/citation-record","integrity":"/paper/2607.29575/integrity","json":"/paper/2607.29575/citation-record.json","paper":"/paper/2607.29575"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T04:22:51.621032Z","title":"The rapid adoption of generative ai,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.29575","last_updated":"2026-07-31T16:02:31Z","snapshot_observed_at":"2026-08-05T23:15:55.566669Z","submitted_at":"2026-07-31T16:02:31Z","title":"SLIM: Saturation-Aware Lightweight Performance Modeling for LLM Serving","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-03T04:22:51.621032Z"},"links":{"citing_paper":"/paper/2607.29575"},"observation_digest":"sha256:d85e7de237533e089a046e95fd0ec06270fed07ce47fc574e594a5d0492de23e","observation_id":"1283f76c-0bbf-4207-80e6-721c88cf4e48","resolution":{"observed_at":"2026-08-03T04:22:51.621032Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T04:22:51.687853Z","title":"The adoption of chatgpt,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.29575","last_updated":"2026-07-31T16:02:31Z","snapshot_observed_at":"2026-08-05T23:15:55.566669Z","submitted_at":"2026-07-31T16:02:31Z","title":"SLIM: Saturation-Aware Lightweight Performance Modeling for LLM Serving","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-03T04:22:51.687853Z"},"links":{"citing_paper":"/paper/2607.29575"},"observation_digest":"sha256:404643356aba82085fcd5a5fb7c745ac2b6a17d655498604fffa28a34e53805b","observation_id":"71750f8e-e65d-4dfc-8fc7-5c8e831faba4","resolution":{"observed_at":"2026-08-03T04:22:51.687853Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T04:22:51.737457Z","title":"Quantifying large language model usage in scientific papers,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.29575","last_updated":"2026-07-31T16:02:31Z","snapshot_observed_at":"2026-08-05T23:15:55.566669Z","submitted_at":"2026-07-31T16:02:31Z","title":"SLIM: Saturation-Aware Lightweight Performance Modeling for LLM Serving","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-03T04:22:51.737457Z"},"links":{"citing_paper":"/paper/2607.29575"},"observation_digest":"sha256:b88d55babb1941d5d48ca3d9419b570afacdaa1e942dadc27566311248c2f5de","observation_id":"5e3aec83-1a24-4388-adde-a9bb58b1fdf0","resolution":{"observed_at":"2026-08-03T04:22:51.737457Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T04:22:51.844613Z","title":"Llumnix: Dynamic scheduling for large language model serving,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.29575","last_updated":"2026-07-31T16:02:31Z","snapshot_observed_at":"2026-08-05T23:15:55.566669Z","submitted_at":"2026-07-31T16:02:31Z","title":"SLIM: Saturation-Aware Lightweight Performance Modeling for LLM Serving","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-03T04:22:51.844613Z"},"links":{"citing_paper":"/paper/2607.29575"},"observation_digest":"sha256:b246f0ab7b28b9160aeb7ac3bdd81ef251f0b2f9ab0abcd8e6aa247b20f753e0","observation_id":"d7a31b72-eb9d-4ed6-8138-bb0287ef2850","resolution":{"observed_at":"2026-08-03T04:22:51.844613Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T04:22:51.935319Z","title":"Sageserve: Optimizing llm serving on cloud data centers with forecast aware auto-scaling,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.29575","last_updated":"2026-07-31T16:02:31Z","snapshot_observed_at":"2026-08-05T23:15:55.566669Z","submitted_at":"2026-07-31T16:02:31Z","title":"SLIM: Saturation-Aware Lightweight Performance Modeling for LLM Serving","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-03T04:22:51.935319Z"},"links":{"citing_paper":"/paper/2607.29575"},"observation_digest":"sha256:acf75b3adab5e56d878344860baa00e78fbcf18979919eccdbca58adda6b169a","observation_id":"a627426c-a4f4-4206-a394-04d805de12ae","resolution":{"observed_at":"2026-08-03T04:22:51.935319Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.04466","last_updated":"2025-06-13T12:20:54Z","snapshot_observed_at":"2026-08-02T03:13:11.983917Z","submitted_at":"2024-10-06T12:42:04Z","title":"Large Language Model Inference Acceleration: A Comprehensive Hardware Perspective","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.04466","snapshot_observed_at":"2026-08-03T04:22:52.032744Z","title":"Large language model inference acceleration: A comprehensive hardware perspective,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.29575","last_updated":"2026-07-31T16:02:31Z","snapshot_observed_at":"2026-08-05T23:15:55.566669Z","submitted_at":"2026-07-31T16:02:31Z","title":"SLIM: Saturation-Aware Lightweight Performance Modeling for LLM Serving","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-03T04:22:52.032744Z"},"links":{"cited_paper":"/paper/2410.04466","citing_paper":"/paper/2607.29575"},"observation_digest":"sha256:235896f4715c3103d1a5cdb235fb1a15507f76c167a0fc5cd211036b6c3bf7a7","observation_id":"dc0fee93-9e4d-46c1-8b28-cfa5f828b8ae","resolution":{"observed_at":"2026-08-03T04:22:52.032744Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T04:22:52.082197Z","title":"Efficiently scaling transformer inference,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.29575","last_updated":"2026-07-31T16:02:31Z","snapshot_observed_at":"2026-08-05T23:15:55.566669Z","submitted_at":"2026-07-31T16:02:31Z","title":"SLIM: Saturation-Aware Lightweight Performance Modeling for LLM Serving","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-03T04:22:52.082197Z"},"links":{"citing_paper":"/paper/2607.29575"},"observation_digest":"sha256:a42b94f596a5bf87296f52d98e6c0ba60d6243987faf61085cace90200e45a2e","observation_id":"3d7e53ee-93d8-47cb-a734-76e6e4e5c880","resolution":{"observed_at":"2026-08-03T04:22:52.082197Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T04:22:52.167267Z","title":"Efficient llm inference: Bandwidth, compute, synchronization, and capacity are all you need,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.29575","last_updated":"2026-07-31T16:02:31Z","snapshot_observed_at":"2026-08-05T23:15:55.566669Z","submitted_at":"2026-07-31T16:02:31Z","title":"SLIM: Saturation-Aware Lightweight Performance Modeling for LLM Serving","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-03T04:22:52.167267Z"},"links":{"citing_paper":"/paper/2607.29575"},"observation_digest":"sha256:53a80379d4f6a4ed732c73c62bef2cfedcc89d1837074091d10eef930f975800","observation_id":"dbd5fb7e-2239-4859-89da-77361ba9beb1","resolution":{"observed_at":"2026-08-03T04:22:52.167267Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T04:22:52.228945Z","title":"Mind the memory gap: Unveiling gpu bot- tlenecks in large-batch llm inference,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.29575","last_updated":"2026-07-31T16:02:31Z","snapshot_observed_at":"2026-08-05T23:15:55.566669Z","submitted_at":"2026-07-31T16:02:31Z","title":"SLIM: Saturation-Aware Lightweight Performance Modeling for LLM Serving","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-03T04:22:52.228945Z"},"links":{"citing_paper":"/paper/2607.29575"},"observation_digest":"sha256:6e0b00200490fbf64a894466ba14cf615097c61a1fd48522fec69afcc8cc32a6","observation_id":"8401ac48-13a1-42e6-96f7-d75a09241a32","resolution":{"observed_at":"2026-08-03T04:22:52.228945Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T04:22:52.312448Z","title":"Llmvisor: A real-time latency attribution model for multi-tenant llm serving,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.29575","last_updated":"2026-07-31T16:02:31Z","snapshot_observed_at":"2026-08-05T23:15:55.566669Z","submitted_at":"2026-07-31T16:02:31Z","title":"SLIM: Saturation-Aware Lightweight Performance Modeling for LLM Serving","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-03T04:22:52.312448Z"},"links":{"citing_paper":"/paper/2607.29575"},"observation_digest":"sha256:44fba6a1e0d359d25eac062fb4010d9c8a02d92220c4d8a33a9d9577ec5b5044","observation_id":"e9be6525-2ed8-4deb-baa1-20852599966e","resolution":{"observed_at":"2026-08-03T04:22:52.312448Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T04:22:52.386461Z","title":"Predicting llm inference latency: A roofline-driven ml method,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.29575","last_updated":"2026-07-31T16:02:31Z","snapshot_observed_at":"2026-08-05T23:15:55.566669Z","submitted_at":"2026-07-31T16:02:31Z","title":"SLIM: Saturation-Aware Lightweight Performance Modeling for LLM Serving","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-03T04:22:52.386461Z"},"links":{"citing_paper":"/paper/2607.29575"},"observation_digest":"sha256:bb0293c779e93de93a64bd24513bc94a80f24766fbff2d15b47cef79f0fbf065","observation_id":"169dd382-9cd3-48bf-8c43-d32d50983289","resolution":{"observed_at":"2026-08-03T04:22:52.386461Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T04:22:52.529800Z","title":"Language mod- els are few-shot learners,","venue":null,"work_id":null,"year":1901},"citing_paper":{"arxiv_id":"2607.29575","last_updated":"2026-07-31T16:02:31Z","snapshot_observed_at":"2026-08-05T23:15:55.566669Z","submitted_at":"2026-07-31T16:02:31Z","title":"SLIM: Saturation-Aware Lightweight Performance Modeling for LLM Serving","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-03T04:22:52.529800Z"},"links":{"citing_paper":"/paper/2607.29575"},"observation_digest":"sha256:9d12e399d125b071fbe62da86d33104b9d7700a181c282540097944493559472","observation_id":"faa6fa12-8937-465d-87bf-2c10a68eb48a","resolution":{"observed_at":"2026-08-03T04:22:52.529800Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2302.13971","last_updated":"2023-02-27T17:11:15Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-02-27T17:11:15Z","title":"LLaMA: Open and Efficient Foundation Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2302.13971","snapshot_observed_at":"2026-08-03T04:22:52.616647Z","title":"Llama: Open and efficient foundation language models,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.29575","last_updated":"2026-07-31T16:02:31Z","snapshot_observed_at":"2026-08-05T23:15:55.566669Z","submitted_at":"2026-07-31T16:02:31Z","title":"SLIM: Saturation-Aware Lightweight Performance Modeling for LLM Serving","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-03T04:22:52.616647Z"},"links":{"cited_paper":"/paper/2302.13971","citing_paper":"/paper/2607.29575"},"observation_digest":"sha256:e84d06d9290d0c90f79c7c31f541a140a7e0f9069fba84f5783c3f28289f1755","observation_id":"06875560-041e-47bd-bca5-4e0a529eaf44","resolution":{"observed_at":"2026-08-03T04:22:52.616647Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T04:22:52.703801Z","title":"Attention is all you need,","venue":null,"work_id":null,"year":2017},"citing_paper":{"arxiv_id":"2607.29575","last_updated":"2026-07-31T16:02:31Z","snapshot_observed_at":"2026-08-05T23:15:55.566669Z","submitted_at":"2026-07-31T16:02:31Z","title":"SLIM: Saturation-Aware Lightweight Performance Modeling for LLM Serving","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-03T04:22:52.703801Z"},"links":{"citing_paper":"/paper/2607.29575"},"observation_digest":"sha256:e6d34884686bf2d4bcbf0f04325d6ac74fc43de6f7ba983ba750744563d151bd","observation_id":"ab551368-fb6b-46ed-b8d7-789c77352651","resolution":{"observed_at":"2026-08-03T04:22:52.703801Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T04:22:52.769107Z","title":"Orca: A distributed serving system for transformer-based generative models,","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2607.29575","last_updated":"2026-07-31T16:02:31Z","snapshot_observed_at":"2026-08-05T23:15:55.566669Z","submitted_at":"2026-07-31T16:02:31Z","title":"SLIM: Saturation-Aware Lightweight Performance Modeling for LLM Serving","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-03T04:22:52.769107Z"},"links":{"citing_paper":"/paper/2607.29575"},"observation_digest":"sha256:3481c02417f75acc83ac630c2e97d4821e02208e2bd0388779c67d6d4cbd4e71","observation_id":"de2897af-d4a1-4f69-8e86-b899daadc481","resolution":{"observed_at":"2026-08-03T04:22:52.769107Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T04:22:52.843887Z","title":"Efficient memory management for large language model serving with pagedattention,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.29575","last_updated":"2026-07-31T16:02:31Z","snapshot_observed_at":"2026-08-05T23:15:55.566669Z","submitted_at":"2026-07-31T16:02:31Z","title":"SLIM: Saturation-Aware Lightweight Performance Modeling for LLM Serving","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-03T04:22:52.843887Z"},"links":{"citing_paper":"/paper/2607.29575"},"observation_digest":"sha256:15add55d9846d82c6b3a8bc44e6a2cf9389e9fa5234b7a542977ba57d3d3b086","observation_id":"56f5eeae-c9e5-433a-b6c4-22b4f42723c9","resolution":{"observed_at":"2026-08-03T04:22:52.843887Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T04:22:52.928323Z","title":"TensorRT-LLM,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.29575","last_updated":"2026-07-31T16:02:31Z","snapshot_observed_at":"2026-08-05T23:15:55.566669Z","submitted_at":"2026-07-31T16:02:31Z","title":"SLIM: Saturation-Aware Lightweight Performance Modeling for LLM Serving","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-03T04:22:52.928323Z"},"links":{"citing_paper":"/paper/2607.29575"},"observation_digest":"sha256:58e78fbfd6ab25860b5a7b6a5833a63d6443e87b33e5bbc704b86b1b993301fc","observation_id":"3aa6c78b-0554-4beb-a73e-df1f537fc2e2","resolution":{"observed_at":"2026-08-03T04:22:52.928323Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T04:22:53.030125Z","title":"DeepSpeed-MII,","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2607.29575","last_updated":"2026-07-31T16:02:31Z","snapshot_observed_at":"2026-08-05T23:15:55.566669Z","submitted_at":"2026-07-31T16:02:31Z","title":"SLIM: Saturation-Aware Lightweight Performance Modeling for LLM Serving","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-03T04:22:53.030125Z"},"links":{"citing_paper":"/paper/2607.29575"},"observation_digest":"sha256:08aaf69c787980011bf19dce4cfdd4a27b0aefce1346436770604b75d1eeeb73","observation_id":"94854dc3-b780-4314-b2a6-ef0ecc871228","resolution":{"observed_at":"2026-08-03T04:22:53.030125Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T04:22:53.117570Z","title":"Slora: Scalable serving of thousands of lora adapters,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.29575","last_updated":"2026-07-31T16:02:31Z","snapshot_observed_at":"2026-08-05T23:15:55.566669Z","submitted_at":"2026-07-31T16:02:31Z","title":"SLIM: Saturation-Aware Lightweight Performance Modeling for LLM Serving","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-03T04:22:53.117570Z"},"links":{"citing_paper":"/paper/2607.29575"},"observation_digest":"sha256:5a997144c698a119a028b18f2b9d89fe908f767bc49624bdbc23b61f4404077f","observation_id":"f142fcee-c690-41b7-a722-d974ca81131e","resolution":{"observed_at":"2026-08-03T04:22:53.117570Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.16363","last_updated":"2024-05-01T20:42:28Z","snapshot_observed_at":"2026-08-01T22:52:35.092898Z","submitted_at":"2024-02-26T07:33:05Z","title":"LLM Inference Unveiled: Survey and Roofline Model Insights","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.16363","snapshot_observed_at":"2026-08-03T04:22:53.202838Z","title":"Llm inference unveiled: Survey and roofline model insights,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.29575","last_updated":"2026-07-31T16:02:31Z","snapshot_observed_at":"2026-08-05T23:15:55.566669Z","submitted_at":"2026-07-31T16:02:31Z","title":"SLIM: Saturation-Aware Lightweight Performance Modeling for LLM Serving","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-03T04:22:53.202838Z"},"links":{"cited_paper":"/paper/2402.16363","citing_paper":"/paper/2607.29575"},"observation_digest":"sha256:e4da40a06a516dc84f6d026e13674086dcc38c4fc2bd8969e899baf8bb68198a","observation_id":"a133c8c9-ab5f-42b6-a05b-570a8d2e5d44","resolution":{"observed_at":"2026-08-03T04:22:53.202838Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T04:22:53.305385Z","title":"Taming{Throughput-Latency}tradeoff in{LLM}inference with{Sarathi-Serve},","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.29575","last_updated":"2026-07-31T16:02:31Z","snapshot_observed_at":"2026-08-05T23:15:55.566669Z","submitted_at":"2026-07-31T16:02:31Z","title":"SLIM: Saturation-Aware Lightweight Performance Modeling for LLM Serving","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-03T04:22:53.305385Z"},"links":{"citing_paper":"/paper/2607.29575"},"observation_digest":"sha256:109c7e358baf0344cdb6e39f88eb4f31e8520a8d8e6cf87fee152b611b543d6b","observation_id":"46ded418-0c69-4282-b2fd-5ca8d11f8edc","resolution":{"observed_at":"2026-08-03T04:22:53.305385Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T04:22:53.428587Z","title":"Fast inference from transform- ers via speculative decoding,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.29575","last_updated":"2026-07-31T16:02:31Z","snapshot_observed_at":"2026-08-05T23:15:55.566669Z","submitted_at":"2026-07-31T16:02:31Z","title":"SLIM: Saturation-Aware Lightweight Performance Modeling for LLM Serving","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-03T04:22:53.428587Z"},"links":{"citing_paper":"/paper/2607.29575"},"observation_digest":"sha256:2489987af9138f609f0cb5239897aed19ba24a64dc33620a2b77befa3f87a56a","observation_id":"e556d816-ff08-40d6-835e-f20286af23c8","resolution":{"observed_at":"2026-08-03T04:22:53.428587Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2302.01318","last_updated":"2023-02-02T18:44:11Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-02-02T18:44:11Z","title":"Accelerating Large Language Model Decoding with Speculative Sampling","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2302.01318","snapshot_observed_at":"2026-08-03T04:22:53.550502Z","title":"Accelerating large language model decoding with speculative sam- pling,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.29575","last_updated":"2026-07-31T16:02:31Z","snapshot_observed_at":"2026-08-05T23:15:55.566669Z","submitted_at":"2026-07-31T16:02:31Z","title":"SLIM: Saturation-Aware Lightweight Performance Modeling for LLM Serving","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-03T04:22:53.550502Z"},"links":{"cited_paper":"/paper/2302.01318","citing_paper":"/paper/2607.29575"},"observation_digest":"sha256:f8f3cc482ffbd1de806d4ef94e5cb674145c80e4aa88359070956dba0d6f35a0","observation_id":"11947e0b-a77a-40a4-9b7f-51eeb4fa4fc2","resolution":{"observed_at":"2026-08-03T04:22:53.550502Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1911.02150","last_updated":"2019-11-06T00:19:05Z","snapshot_observed_at":"2026-07-06T08:35:01.386074Z","submitted_at":"2019-11-06T00:19:05Z","title":"Fast Transformer Decoding: One Write-Head is All You Need","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1911.02150","snapshot_observed_at":"2026-08-03T04:22:53.668526Z","title":"Fast transformer decoding: One write-head is all you need,","venue":null,"work_id":null,"year":1911},"citing_paper":{"arxiv_id":"2607.29575","last_updated":"2026-07-31T16:02:31Z","snapshot_observed_at":"2026-08-05T23:15:55.566669Z","submitted_at":"2026-07-31T16:02:31Z","title":"SLIM: Saturation-Aware Lightweight Performance Modeling for LLM Serving","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-03T04:22:53.668526Z"},"links":{"cited_paper":"/paper/1911.02150","citing_paper":"/paper/2607.29575"},"observation_digest":"sha256:a65b6f857d8430f9501c04cfaff0660b9e98c1ca28482049068a1abb35a880e3","observation_id":"f0c8d29b-ce4a-4423-b2d6-fc9afdaae304","resolution":{"observed_at":"2026-08-03T04:22:53.668526Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.13245","last_updated":"2023-12-23T17:55:11Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-22T17:16:38Z","title":"GQA: Training Generalized Multi-Query Transformer Models from Multi-Head Checkpoints","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.13245","snapshot_observed_at":"2026-08-03T04:22:53.767380Z","title":"Gqa: Training generalized multi-query transformer models from multi-head checkpoints,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.29575","last_updated":"2026-07-31T16:02:31Z","snapshot_observed_at":"2026-08-05T23:15:55.566669Z","submitted_at":"2026-07-31T16:02:31Z","title":"SLIM: Saturation-Aware Lightweight Performance Modeling for LLM Serving","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-03T04:22:53.767380Z"},"links":{"cited_paper":"/paper/2305.13245","citing_paper":"/paper/2607.29575"},"observation_digest":"sha256:3f3df08c303b1cef6a6365b6a495cfbae708160903a618b9c7a99dd609fa36c8","observation_id":"4fff4c06-15c0-4d6c-82a7-e8a0c5e61c97","resolution":{"observed_at":"2026-08-03T04:22:53.767380Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T04:22:53.847735Z","title":"Awq: Activation-aware weight quanti- zation for on-device llm compression and acceleration,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.29575","last_updated":"2026-07-31T16:02:31Z","snapshot_observed_at":"2026-08-05T23:15:55.566669Z","submitted_at":"2026-07-31T16:02:31Z","title":"SLIM: Saturation-Aware Lightweight Performance Modeling for LLM Serving","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-03T04:22:53.847735Z"},"links":{"citing_paper":"/paper/2607.29575"},"observation_digest":"sha256:f110c578efae5ff875df82ce59cd2d2f85406419bfd206d2bed3ea285a6e74a0","observation_id":"8a9d34a5-5e25-41b3-b4f1-f9dde91eb23f","resolution":{"observed_at":"2026-08-03T04:22:53.847735Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2210.17323","last_updated":"2023-03-22T13:10:47Z","snapshot_observed_at":"2026-08-07T08:38:54.025062Z","submitted_at":"2022-10-31T13:42:40Z","title":"GPTQ: Accurate Post-Training Quantization for Generative Pre-trained Transformers","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2210.17323","snapshot_observed_at":"2026-08-03T04:22:53.915456Z","title":"Gptq: Accurate post-training quantization for generative pre-trained transformers,","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2607.29575","last_updated":"2026-07-31T16:02:31Z","snapshot_observed_at":"2026-08-05T23:15:55.566669Z","submitted_at":"2026-07-31T16:02:31Z","title":"SLIM: Saturation-Aware Lightweight Performance Modeling for LLM Serving","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-03T04:22:53.915456Z"},"links":{"cited_paper":"/paper/2210.17323","citing_paper":"/paper/2607.29575"},"observation_digest":"sha256:13feaea1bea81ae0af7e39aaa7d688c8eef9e76c884f3995c335cfe2d168ce17","observation_id":"6f35459a-2052-4824-bc59-7fc008a098fc","resolution":{"observed_at":"2026-08-03T04:22:53.915456Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T04:22:54.024548Z","title":"Vidur: A large-scale simulation framework for llm inference,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.29575","last_updated":"2026-07-31T16:02:31Z","snapshot_observed_at":"2026-08-05T23:15:55.566669Z","submitted_at":"2026-07-31T16:02:31Z","title":"SLIM: Saturation-Aware Lightweight Performance Modeling for LLM Serving","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-03T04:22:54.024548Z"},"links":{"citing_paper":"/paper/2607.29575"},"observation_digest":"sha256:4b4ff08b89a8bcf70570758e549ea64b392ffb38aa7d4d799287f8ac6fa212de","observation_id":"ec390e4f-1a26-45ca-b8ef-c96e3928b659","resolution":{"observed_at":"2026-08-03T04:22:54.024548Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.01698","last_updated":"2025-05-15T02:46:53Z","snapshot_observed_at":"2026-08-05T11:19:39.261458Z","submitted_at":"2024-06-03T18:00:50Z","title":"Demystifying AI Platform Design for Distributed Inference of Next-Generation LLM models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.01698","snapshot_observed_at":"2026-08-03T04:22:54.136193Z","title":"Demystifying ai platform design for distributed inference of next-generation llm models,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.29575","last_updated":"2026-07-31T16:02:31Z","snapshot_observed_at":"2026-08-05T23:15:55.566669Z","submitted_at":"2026-07-31T16:02:31Z","title":"SLIM: Saturation-Aware Lightweight Performance Modeling for LLM Serving","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-03T04:22:54.136193Z"},"links":{"cited_paper":"/paper/2406.01698","citing_paper":"/paper/2607.29575"},"observation_digest":"sha256:8eb118c07a9d51903c93a491caa0c00ab48dc059bc2331e9486e0d82640f93ad","observation_id":"9474e2ac-8847-448d-a2c5-33784756ba3a","resolution":{"observed_at":"2026-08-03T04:22:54.136193Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T04:22:54.209163Z","title":"Llmcompass: Enabling efficient hardware design for large language model inference,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.29575","last_updated":"2026-07-31T16:02:31Z","snapshot_observed_at":"2026-08-05T23:15:55.566669Z","submitted_at":"2026-07-31T16:02:31Z","title":"SLIM: Saturation-Aware Lightweight Performance Modeling for LLM Serving","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-03T04:22:54.209163Z"},"links":{"citing_paper":"/paper/2607.29575"},"observation_digest":"sha256:d4e101bd2141f80f4311428de6e7e93c6bc2acc99e283b953a1a28f2173d49bd","observation_id":"e3a09dfd-6bd0-4c5c-b9c0-dad1f0f4702a","resolution":{"observed_at":"2026-08-03T04:22:54.209163Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T04:22:54.321060Z","title":"Amali: An analytical model for accurately modeling llm inference on modern gpus,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.29575","last_updated":"2026-07-31T16:02:31Z","snapshot_observed_at":"2026-08-05T23:15:55.566669Z","submitted_at":"2026-07-31T16:02:31Z","title":"SLIM: Saturation-Aware Lightweight Performance Modeling for LLM Serving","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-03T04:22:54.321060Z"},"links":{"citing_paper":"/paper/2607.29575"},"observation_digest":"sha256:8a92944d2d88219390cd2b0418d3789465f191df4ba6c2d25cbda3e51ecbf760","observation_id":"8d956b91-494c-4617-83d1-4b357831ad48","resolution":{"observed_at":"2026-08-03T04:22:54.321060Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T04:22:54.393342Z","title":"Fairness in serving large language models,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.29575","last_updated":"2026-07-31T16:02:31Z","snapshot_observed_at":"2026-08-05T23:15:55.566669Z","submitted_at":"2026-07-31T16:02:31Z","title":"SLIM: Saturation-Aware Lightweight Performance Modeling for LLM Serving","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-03T04:22:54.393342Z"},"links":{"citing_paper":"/paper/2607.29575"},"observation_digest":"sha256:0963e3b25b3041ac2cfb0877a518e6edf5012005faeb59b4a7348d74d36e6bf2","observation_id":"46db8890-3e1f-4963-9f18-6cc14c0f8fef","resolution":{"observed_at":"2026-08-03T04:22:54.393342Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T04:22:54.458412Z","title":"Clean sharegpt dataset,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.29575","last_updated":"2026-07-31T16:02:31Z","snapshot_observed_at":"2026-08-05T23:15:55.566669Z","submitted_at":"2026-07-31T16:02:31Z","title":"SLIM: Saturation-Aware Lightweight Performance Modeling for LLM Serving","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-03T04:22:54.458412Z"},"links":{"citing_paper":"/paper/2607.29575"},"observation_digest":"sha256:3101426700a8da75e96e9e6fb0173b9ea6a646a9333feb7d956832976f647518","observation_id":"ef1c6471-f486-4f1d-a3b2-dc42413c0c53","resolution":{"observed_at":"2026-08-03T04:22:54.458412Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.06825","last_updated":"2023-10-10T17:54:58Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-10-10T17:54:58Z","title":"Mistral 7B","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.06825","snapshot_observed_at":"2026-08-03T04:22:54.533081Z","title":"Mistral 7b,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.29575","last_updated":"2026-07-31T16:02:31Z","snapshot_observed_at":"2026-08-05T23:15:55.566669Z","submitted_at":"2026-07-31T16:02:31Z","title":"SLIM: Saturation-Aware Lightweight Performance Modeling for LLM Serving","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-03T04:22:54.533081Z"},"links":{"cited_paper":"/paper/2310.06825","citing_paper":"/paper/2607.29575"},"observation_digest":"sha256:81af38a33d855a4ac5b3d24903aab8b3c235976a1abb9294da35a1c7cb85c5bf","observation_id":"6b314849-88b5-42a6-910a-381b7b4b23f0","resolution":{"observed_at":"2026-08-03T04:22:54.533081Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.04324","last_updated":"2024-05-07T13:50:40Z","snapshot_observed_at":"2026-07-06T18:11:01.734775Z","submitted_at":"2024-05-07T13:50:40Z","title":"Granite Code Models: A Family of Open Foundation Models for Code Intelligence","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.04324","snapshot_observed_at":"2026-08-03T04:22:54.615902Z","title":"Granite code models: A family of open foundation models for code intelligence,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.29575","last_updated":"2026-07-31T16:02:31Z","snapshot_observed_at":"2026-08-05T23:15:55.566669Z","submitted_at":"2026-07-31T16:02:31Z","title":"SLIM: Saturation-Aware Lightweight Performance Modeling for LLM Serving","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-03T04:22:54.615902Z"},"links":{"cited_paper":"/paper/2405.04324","citing_paper":"/paper/2607.29575"},"observation_digest":"sha256:8b93d9ca3dc463c6d5a12aae4a7de7cd65e29480c6a6474bc6c6f1194cb7b820","observation_id":"a2a6e250-6339-4122-b15f-b4bac850f715","resolution":{"observed_at":"2026-08-03T04:22:54.615902Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T04:22:54.707629Z","title":"Opt: Open pre-trained transformer language models,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.29575","last_updated":"2026-07-31T16:02:31Z","snapshot_observed_at":"2026-08-05T23:15:55.566669Z","submitted_at":"2026-07-31T16:02:31Z","title":"SLIM: Saturation-Aware Lightweight Performance Modeling for LLM Serving","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-03T04:22:54.707629Z"},"links":{"citing_paper":"/paper/2607.29575"},"observation_digest":"sha256:5852cf9f54af82c4d7037b5c38b36bf4c28ae29ddf846294ad84843a16da8dd5","observation_id":"ab58eaea-b3aa-4790-bff5-1bc003501ea7","resolution":{"observed_at":"2026-08-03T04:22:54.707629Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.10671","last_updated":"2024-09-10T13:25:53Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-07-15T12:35:42Z","title":"Qwen2 Technical Report","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.10671","snapshot_observed_at":"2026-08-03T04:22:54.849854Z","title":"Qwen2 technical report,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.29575","last_updated":"2026-07-31T16:02:31Z","snapshot_observed_at":"2026-08-05T23:15:55.566669Z","submitted_at":"2026-07-31T16:02:31Z","title":"SLIM: Saturation-Aware Lightweight Performance Modeling for LLM Serving","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-03T04:22:54.849854Z"},"links":{"cited_paper":"/paper/2407.10671","citing_paper":"/paper/2607.29575"},"observation_digest":"sha256:1126a57b4288a134fd43dc267736c78831d23093d9eae7d4718287cd8b0c1b20","observation_id":"543d0be1-ba38-436c-9a08-fd300a83dc8a","resolution":{"observed_at":"2026-08-03T04:22:54.849854Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-03T04:22:54.961186Z","title":"Ai and memory wall,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.29575","last_updated":"2026-07-31T16:02:31Z","snapshot_observed_at":"2026-08-05T23:15:55.566669Z","submitted_at":"2026-07-31T16:02:31Z","title":"SLIM: Saturation-Aware Lightweight Performance Modeling for LLM Serving","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-03T04:22:54.961186Z"},"links":{"citing_paper":"/paper/2607.29575"},"observation_digest":"sha256:9b50668c97077562b4dfc11e06f7bb83a73ac775b386b826bddf298e00386b1c","observation_id":"c589949a-a181-4c18-b42c-a44f50ab1084","resolution":{"observed_at":"2026-08-03T04:22:54.961186Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2205.01068","last_updated":"2022-06-21T17:04:40Z","snapshot_observed_at":"2026-08-06T03:13:37.403059Z","submitted_at":"2022-05-02T17:49:50Z","title":"OPT: Open Pre-trained Transformer Language Models","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2205.01068","snapshot_observed_at":"2026-08-03T04:22:54.775813Z","title":"Available: https://arxiv.org/abs/2205.01068","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.29575","last_updated":"2026-07-31T16:02:31Z","snapshot_observed_at":"2026-08-05T23:15:55.566669Z","submitted_at":"2026-07-31T16:02:31Z","title":"SLIM: Saturation-Aware Lightweight Performance Modeling for LLM Serving","version":1},"reference_index":2022,"source":"pdf_text","source_observed_at":"2026-08-03T04:22:54.775813Z"},"links":{"cited_paper":"/paper/2205.01068","citing_paper":"/paper/2607.29575"},"observation_digest":"sha256:48304ddaf7f0fc92dd8fa9b923a985d896fc52809d51b28c7e812a435c653cef","observation_id":"faa8d4de-c4d2-4dc7-919c-4814191e17ec","resolution":{"observed_at":"2026-08-03T04:22:54.775813Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2607.29575","last_updated":"2026-07-31T16:02:31Z","latest_version":1,"primary_category":"cs.DC","snapshot_observed_at":"2026-08-05T23:15:55.566669Z","submitted_at":"2026-07-31T16:02:31Z","title":"SLIM: Saturation-Aware Lightweight Performance Modeling for LLM Serving"},"reference_resolution":{"displayed":39,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":39,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":39},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"thesis":"As of 7 August 2026, this Paper Citation Record lists 39 of 39 outbound references and 0 inbound Pith citation observations for arXiv:2607.29575."}