{"as_of":"2026-08-05T11:39:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:660cbef2f64c315543abfc309eab17910f9b10490efbec5bafdcb8785f8bde64","coverage":[{"denominator":24,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":24,"source":"paper_references, paper_reference_links","source_observed_at":"2026-05-15T00:37:08.671539Z","state":"measured"},{"denominator":24,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":24,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-05T06:32:48.257954+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2605.00831/citation-record","integrity":"/paper/2605.00831/integrity","json":"/paper/2605.00831/citation-record.json","paper":"/paper/2605.00831"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2308.16369","last_updated":"2023-08-31T00:03:02Z","snapshot_observed_at":"2026-08-01T15:59:13.579598Z","submitted_at":"2023-08-31T00:03:02Z","title":"SARATHI: Efficient LLM Inference by Piggybacking Decodes with Chunked Prefills","version":1},"cited_work":{"arxiv_id":"2308.16369","doi":"10.48550/arxiv.2308.16369","metadata_source":"pith","pith_arxiv_id":"2308.16369","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SARATHI: Efficient LLM Inference by Piggybacking Decodes with Chunked Prefills","venue":"cs.LG","work_id":"3dbdd757-ca01-436f-acfd-12ffcd6f64c6","year":2023},"citing_paper":{"arxiv_id":"2605.00831","last_updated":"2026-03-26T13:27:57Z","snapshot_observed_at":"2026-08-02T06:03:58.648365Z","submitted_at":"2026-03-26T13:27:57Z","title":"GhostServe: A Lightweight Checkpointing System in the Shadow for Fault-Tolerant LLM Serving","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-15T00:37:08.671539Z"},"links":{"cited_paper":"/paper/2308.16369","citing_paper":"/paper/2605.00831"},"observation_digest":"sha256:307bf1ae10211dfe6e5de28dbee1191f91dd1073755bb551c36899cb809707dc","observation_id":"1f32c318-668d-4544-9bcd-c918306c2685","resolution":{"observed_at":"2026-05-16T06:31:47.535686Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2409.17264","doi":"10.48550/arxiv.2409.17264","metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"2025.No Request Left Behind: Tackling Heterogeneity in Long-Context LLM Inference with Medha","venue":"arXiv (Cornell University)","work_id":"3eb4a0fb-50a0-4b35-a96e-ac065a397497","year":2024},"citing_paper":{"arxiv_id":"2605.00831","last_updated":"2026-03-26T13:27:57Z","snapshot_observed_at":"2026-08-02T06:03:58.648365Z","submitted_at":"2026-03-26T13:27:57Z","title":"GhostServe: A Lightweight Checkpointing System in the Shadow for Fault-Tolerant LLM Serving","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-15T00:37:08.671539Z"},"links":{"citing_paper":"/paper/2605.00831"},"observation_digest":"sha256:07ac8c5337d43a437d2a44d4c696570de27ab56a4aca98e1fb21b59163f31e85","observation_id":"40cf2f55-f826-430b-b478-0bb215411241","resolution":{"observed_at":"2026-05-15T00:38:23.090087Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"K., Janakiraman, R., and Xu, L","venue":null,"work_id":"1042c8d1-c130-4086-9a13-ecf0664069ff","year":2005},"citing_paper":{"arxiv_id":"2605.00831","last_updated":"2026-03-26T13:27:57Z","snapshot_observed_at":"2026-08-02T06:03:58.648365Z","submitted_at":"2026-03-26T13:27:57Z","title":"GhostServe: A Lightweight Checkpointing System in the Shadow for Fault-Tolerant LLM Serving","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-15T00:37:08.671539Z"},"links":{"citing_paper":"/paper/2605.00831"},"observation_digest":"sha256:96782b5652012b73af11087aebeca9ba24e47b6c6333a43d27591ee0654597a6","observation_id":"4f336952-cfe7-480a-9cf2-fc50bb54a60a","resolution":{"observed_at":"2026-05-15T00:38:23.691453Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-08T09:34:48.543823Z","title":"D., Dhariwal, P., Neelakantan, A., Shyam, P., Sastry, G., Askell, A., et al","venue":null,"work_id":"9806adeb-7378-4bee-a184-3e98c89988dd","year":1901},"citing_paper":{"arxiv_id":"2605.00831","last_updated":"2026-03-26T13:27:57Z","snapshot_observed_at":"2026-08-02T06:03:58.648365Z","submitted_at":"2026-03-26T13:27:57Z","title":"GhostServe: A Lightweight Checkpointing System in the Shadow for Fault-Tolerant LLM Serving","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-15T00:37:08.671539Z"},"links":{"citing_paper":"/paper/2605.00831"},"observation_digest":"sha256:bd39eb50c66247dde48bddcedcb7256d556d9cc7f38fe51b5fe19101081dcfd1","observation_id":"f189bdce-0ca6-4c60-a276-1464c4d55027","resolution":{"observed_at":"2026-05-15T00:38:23.693289Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.15465","last_updated":"2025-04-21T22:21:39Z","snapshot_observed_at":"2026-07-06T21:12:46.642950Z","submitted_at":"2025-04-21T22:21:39Z","title":"LithOS: An Operating System for Efficient Machine Learning on GPUs","version":1},"cited_work":{"arxiv_id":"2504.15465","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2504.15465","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"H., Zhang, B., Solomon, E","venue":null,"work_id":"bc0cc9f0-1098-44f1-a4c5-746c5ad95906","year":null},"citing_paper":{"arxiv_id":"2605.00831","last_updated":"2026-03-26T13:27:57Z","snapshot_observed_at":"2026-08-02T06:03:58.648365Z","submitted_at":"2026-03-26T13:27:57Z","title":"GhostServe: A Lightweight Checkpointing System in the Shadow for Fault-Tolerant LLM Serving","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-05-15T00:37:08.671539Z"},"links":{"cited_paper":"/paper/2504.15465","citing_paper":"/paper/2605.00831"},"observation_digest":"sha256:b237bca831ff2c8d216580942e1d76cf76cb06f68e1097130959fc3b7261ac03","observation_id":"03632557-c26b-4c89-a771-071cdba79daf","resolution":{"observed_at":"2026-05-15T00:38:23.080048Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.08691","last_updated":"2023-07-17T17:50:36Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-07-17T17:50:36Z","title":"FlashAttention-2: Faster Attention with Better Parallelism and Work Partitioning","version":1},"cited_work":{"arxiv_id":"2307.08691","doi":"10.48550/arxiv.2307.08691","metadata_source":"pith","pith_arxiv_id":"2307.08691","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"FlashAttention-2: Faster Attention with Better Parallelism and Work Partitioning","venue":"cs.LG","work_id":"fff3953b-5efb-4753-bee4-002f59995810","year":2023},"citing_paper":{"arxiv_id":"2605.00831","last_updated":"2026-03-26T13:27:57Z","snapshot_observed_at":"2026-08-02T06:03:58.648365Z","submitted_at":"2026-03-26T13:27:57Z","title":"GhostServe: A Lightweight Checkpointing System in the Shadow for Fault-Tolerant LLM Serving","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-05-15T00:37:08.671539Z"},"links":{"cited_paper":"/paper/2307.08691","citing_paper":"/paper/2605.00831"},"observation_digest":"sha256:20a886925babc0f3b58b260eb06f098b567eecdc9d4aeab1a548665943eb5494","observation_id":"ba6c971b-44b2-4d65-9383-188482e59c92","resolution":{"observed_at":"2026-05-15T00:38:23.086652Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-07-13T15:50:18.243793+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-13T15:50:18.243793+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"G., and Yang, J","venue":null,"work_id":"ba083b62-b7c7-4864-b1f0-754bb9f76c5c","year":2021},"citing_paper":{"arxiv_id":"2605.00831","last_updated":"2026-03-26T13:27:57Z","snapshot_observed_at":"2026-08-02T06:03:58.648365Z","submitted_at":"2026-03-26T13:27:57Z","title":"GhostServe: A Lightweight Checkpointing System in the Shadow for Fault-Tolerant LLM Serving","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-05-15T00:37:08.671539Z"},"links":{"citing_paper":"/paper/2605.00831"},"observation_digest":"sha256:c4a7081327cf3f2b8428524bbc0d315c85e03151aa3ef13e254925e7697f7445","observation_id":"7fbb0742-452e-4d74-a18e-27b87f258025","resolution":{"observed_at":"2026-05-15T00:38:23.695085Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12948","last_updated":"2026-01-04T03:57:36Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-01-22T15:19:35Z","title":"DeepSeek-R1: Incentivizing Reasoning Capability in LLMs via Reinforcement Learning","version":2},"cited_work":{"arxiv_id":"2501.12948","doi":"10.1016/j.artmed.2024.103001","metadata_source":"pith","pith_arxiv_id":"2501.12948","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"DeepSeek-R1: Incentivizing Reasoning Capability in LLMs via Reinforcement Learning","venue":"cs.CL","work_id":"e6b75ad5-2877-4168-97c8-710407094d20","year":2025},"citing_paper":{"arxiv_id":"2605.00831","last_updated":"2026-03-26T13:27:57Z","snapshot_observed_at":"2026-08-02T06:03:58.648365Z","submitted_at":"2026-03-26T13:27:57Z","title":"GhostServe: A Lightweight Checkpointing System in the Shadow for Fault-Tolerant LLM Serving","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-05-15T00:37:08.671539Z"},"links":{"cited_paper":"/paper/2501.12948","citing_paper":"/paper/2605.00831"},"observation_digest":"sha256:ba617167e1612d72e2a09fdb46e3b2b084b810d7f0adda3da03bb3a1e1870e27","observation_id":"5e329564-c35a-48a8-a297-97cca508cf6b","resolution":{"observed_at":"2026-05-15T00:38:23.137446Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.05066","last_updated":"2026-05-11T07:44:47Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-03-07T01:11:39Z","title":"Capacity-Aware Inference: Mitigating the Straggler Effect in Mixture of Experts","version":5},"cited_work":{"arxiv_id":"2503.05066","doi":null,"metadata_source":"pith","pith_arxiv_id":"2503.05066","snapshot_observed_at":"2026-07-04T05:09:36.842254Z","title":"Capacity-Aware Inference: Mitigating the Straggler Effect in Mixture of Experts","venue":"cs.LG","work_id":"4e6941fb-81fb-4660-9150-fefeebc3409a","year":2025},"citing_paper":{"arxiv_id":"2605.00831","last_updated":"2026-03-26T13:27:57Z","snapshot_observed_at":"2026-08-02T06:03:58.648365Z","submitted_at":"2026-03-26T13:27:57Z","title":"GhostServe: A Lightweight Checkpointing System in the Shadow for Fault-Tolerant LLM Serving","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-05-15T00:37:08.671539Z"},"links":{"cited_paper":"/paper/2503.05066","citing_paper":"/paper/2605.00831"},"observation_digest":"sha256:de902fff7f25bafd8f9712849d919ee5300038a4cbfe460cfef62c6879503d99","observation_id":"5f1a7cbd-251c-4db1-9a9d-330552fe9476","resolution":{"observed_at":"2026-05-15T00:38:23.083416Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2203.15556","last_updated":"2022-03-29T13:38:03Z","snapshot_observed_at":"2026-07-06T12:54:11.616335Z","submitted_at":"2022-03-29T13:38:03Z","title":"Training Compute-Optimal Large Language Models","version":1},"cited_work":{"arxiv_id":"2203.15556","doi":"10.1098/rsta.2024.0522","metadata_source":"pith","pith_arxiv_id":"2203.15556","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Training Compute-Optimal Large Language Models","venue":"cs.CL","work_id":"b2faf28d-86b7-429c-bc42-469458efc246","year":2022},"citing_paper":{"arxiv_id":"2605.00831","last_updated":"2026-03-26T13:27:57Z","snapshot_observed_at":"2026-08-02T06:03:58.648365Z","submitted_at":"2026-03-26T13:27:57Z","title":"GhostServe: A Lightweight Checkpointing System in the Shadow for Fault-Tolerant LLM Serving","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-05-15T00:37:08.671539Z"},"links":{"cited_paper":"/paper/2203.15556","citing_paper":"/paper/2605.00831"},"observation_digest":"sha256:6b379c4d803a313b20885edc1048ccce2a562b4ac7002fec9d4afa48a61b6bca","observation_id":"1115cdd8-ca82-4100-bfc7-8462b890676d","resolution":{"observed_at":"2026-05-15T00:38:23.114487Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.00722","last_updated":"2025-06-05T07:54:05Z","snapshot_observed_at":"2026-07-06T20:29:45.369233Z","submitted_at":"2025-02-02T08:44:43Z","title":"Demystifying Cost-Efficiency in LLM Serving over Heterogeneous GPUs","version":2},"cited_work":{"arxiv_id":"2502.00722","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2502.00722","snapshot_observed_at":"2026-07-04T02:39:25.562124Z","title":"Demystifying cost-efficiency in llm serving over heterogeneous gpus","venue":null,"work_id":"a6e25c52-ee2f-47b7-a356-3a5806580152","year":2025},"citing_paper":{"arxiv_id":"2605.00831","last_updated":"2026-03-26T13:27:57Z","snapshot_observed_at":"2026-08-02T06:03:58.648365Z","submitted_at":"2026-03-26T13:27:57Z","title":"GhostServe: A Lightweight Checkpointing System in the Shadow for Fault-Tolerant LLM Serving","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-05-15T00:37:08.671539Z"},"links":{"cited_paper":"/paper/2502.00722","citing_paper":"/paper/2605.00831"},"observation_digest":"sha256:7d1bc46d047318632be2360f40607e8b287756036da086d000044f195e0b3040","observation_id":"50cd5564-0d7d-473b-9401-79f63bd87680","resolution":{"observed_at":"2026-05-15T00:38:23.108123Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2001.08361","last_updated":"2020-01-23T03:59:20Z","snapshot_observed_at":"2026-07-06T08:52:12.656082Z","submitted_at":"2020-01-23T03:59:20Z","title":"Scaling Laws for Neural Language Models","version":1},"cited_work":{"arxiv_id":"2001.08361","doi":"10.1145/3616855.3635845","metadata_source":"pith","pith_arxiv_id":"2001.08361","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Scaling Laws for Neural Language Models","venue":"cs.LG","work_id":"b7dd8749-9c45-4977-ab9b-64478dce1ae8","year":2020},"citing_paper":{"arxiv_id":"2605.00831","last_updated":"2026-03-26T13:27:57Z","snapshot_observed_at":"2026-08-02T06:03:58.648365Z","submitted_at":"2026-03-26T13:27:57Z","title":"GhostServe: A Lightweight Checkpointing System in the Shadow for Fault-Tolerant LLM Serving","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-05-15T00:37:08.671539Z"},"links":{"cited_paper":"/paper/2001.08361","citing_paper":"/paper/2605.00831"},"observation_digest":"sha256:849295c3ebf9cffb0b0fdf4509e4110f7053f10ad64aa8b8fc9fe8dd35fe40d9","observation_id":"bb503c89-5bcb-4c9e-91b2-9976bb34ac6f","resolution":{"observed_at":"2026-05-15T00:38:23.100128Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Revisiting reliability in large-scale machine learning research clusters.2025 IEEE International Symposium on High Performance Computer Architecture (HPCA), pp","venue":null,"work_id":"5b0f968c-ad5c-4066-be33-d47e6310d422","year":2025},"citing_paper":{"arxiv_id":"2605.00831","last_updated":"2026-03-26T13:27:57Z","snapshot_observed_at":"2026-08-02T06:03:58.648365Z","submitted_at":"2026-03-26T13:27:57Z","title":"GhostServe: A Lightweight Checkpointing System in the Shadow for Fault-Tolerant LLM Serving","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-05-15T00:37:08.671539Z"},"links":{"citing_paper":"/paper/2605.00831"},"observation_digest":"sha256:1531dfce41a5434be2300535d81ec533c28cd8b70dba95764944c932dcce80a2","observation_id":"7221dabb-4205-4175-9750-75eb80846d88","resolution":{"observed_at":"2026-05-15T00:38:23.687864Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.05713","last_updated":"2025-05-12T17:52:35Z","snapshot_observed_at":"2026-07-06T21:21:14.988474Z","submitted_at":"2025-05-09T01:24:24Z","title":"Understanding Stragglers in Large Model Training Using What-if Analysis","version":2},"cited_work":{"arxiv_id":"2505.05713","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2505.05713","snapshot_observed_at":"2026-07-01T13:35:46.651338Z","title":"Understanding stragglers in large model training using what-if analysis","venue":null,"work_id":"d6b07ca0-b548-45db-bd5f-8733d18bd11a","year":2025},"citing_paper":{"arxiv_id":"2605.00831","last_updated":"2026-03-26T13:27:57Z","snapshot_observed_at":"2026-08-02T06:03:58.648365Z","submitted_at":"2026-03-26T13:27:57Z","title":"GhostServe: A Lightweight Checkpointing System in the Shadow for Fault-Tolerant LLM Serving","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-05-15T00:37:08.671539Z"},"links":{"cited_paper":"/paper/2505.05713","citing_paper":"/paper/2605.00831"},"observation_digest":"sha256:fa33a82b824531e12d5ad702021f5b70f7023b72613cb801670d24cfeacd65a2","observation_id":"d3ff529d-b3d9-46e5-8ea7-b8a05c71a8b9","resolution":{"observed_at":"2026-05-15T00:38:23.118021Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2511.07424","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Enhancing relia- bility in ai inference services: An empirical study on real production incidents.ArXiv, abs/2511.07424","venue":null,"work_id":"f8a5194c-b5cf-485e-8b15-17078c40072d","year":null},"citing_paper":{"arxiv_id":"2605.00831","last_updated":"2026-03-26T13:27:57Z","snapshot_observed_at":"2026-08-02T06:03:58.648365Z","submitted_at":"2026-03-26T13:27:57Z","title":"GhostServe: A Lightweight Checkpointing System in the Shadow for Fault-Tolerant LLM Serving","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-05-15T00:37:08.671539Z"},"links":{"citing_paper":"/paper/2605.00831"},"observation_digest":"sha256:fbab373fefb68d9140667d0cc81cf8cc659add9047174ad356fb1b8da23c63b1","observation_id":"889ee8a7-05b6-4da5-8069-68d56240f472","resolution":{"observed_at":"2026-05-15T00:38:23.104043Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2602.00277","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-03T17:28:43.923244Z","title":"Salpekar, R","venue":null,"work_id":"73250eae-ea98-4123-bd75-49d784c77b69","year":2026},"citing_paper":{"arxiv_id":"2605.00831","last_updated":"2026-03-26T13:27:57Z","snapshot_observed_at":"2026-08-02T06:03:58.648365Z","submitted_at":"2026-03-26T13:27:57Z","title":"GhostServe: A Lightweight Checkpointing System in the Shadow for Fault-Tolerant LLM Serving","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-05-15T00:37:08.671539Z"},"links":{"citing_paper":"/paper/2605.00831"},"observation_digest":"sha256:649ffc2dd80a5cc99f5656c1d5510d175bb5bfc86b24b6d10c4df8ff16cdcc96","observation_id":"1283926b-249c-4c8a-a9d8-13a88c0ba556","resolution":{"observed_at":"2026-05-15T00:38:23.077148Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1909.08053","last_updated":"2020-03-13T23:45:18Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2019-09-17T19:42:54Z","title":"Megatron-LM: Training Multi-Billion Parameter Language Models Using Model Parallelism","version":4},"cited_work":{"arxiv_id":"1909.08053","doi":"10.48550/arxiv.1909.08053","metadata_source":"pith","pith_arxiv_id":"1909.08053","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Megatron-LM: Training Multi-Billion Parameter Language Models Using Model Parallelism","venue":"cs.CL","work_id":"c888e6d1-0b1d-43d6-9ef5-f0912a0efa1b","year":2019},"citing_paper":{"arxiv_id":"2605.00831","last_updated":"2026-03-26T13:27:57Z","snapshot_observed_at":"2026-08-02T06:03:58.648365Z","submitted_at":"2026-03-26T13:27:57Z","title":"GhostServe: A Lightweight Checkpointing System in the Shadow for Fault-Tolerant LLM Serving","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-05-15T00:37:08.671539Z"},"links":{"cited_paper":"/paper/1909.08053","citing_paper":"/paper/2605.00831"},"observation_digest":"sha256:f8986f9c7436e0c22bc759d77b17a87b8c2b3aa8fcd75f4eb34c2543e973c0b3","observation_id":"3d6f1dc7-4be7-4ede-a732-8a8f007b9148","resolution":{"observed_at":"2026-05-15T00:38:23.130518Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-07-09T10:48:33.392193+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-09T10:48:33.392193+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"2024 crowdstrike-related it outages","venue":null,"work_id":"ca8aeb64-ad3b-40c9-a518-a6079155b300","year":2024},"citing_paper":{"arxiv_id":"2605.00831","last_updated":"2026-03-26T13:27:57Z","snapshot_observed_at":"2026-08-02T06:03:58.648365Z","submitted_at":"2026-03-26T13:27:57Z","title":"GhostServe: A Lightweight Checkpointing System in the Shadow for Fault-Tolerant LLM Serving","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-05-15T00:37:08.671539Z"},"links":{"citing_paper":"/paper/2605.00831"},"observation_digest":"sha256:a9b13d74246088124aa5e5ffe589a08b3741d210ff260a5598a97b4c287bf4fc","observation_id":"854b60c6-4938-49f3-a83d-f527498fb772","resolution":{"observed_at":"2026-05-15T00:38:23.689769Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1910.03771","last_updated":"2020-07-14T03:42:34Z","snapshot_observed_at":"2026-07-06T08:27:58.343233Z","submitted_at":"2019-10-09T03:23:22Z","title":"HuggingFace's Transformers: State-of-the-art Natural Language Processing","version":5},"cited_work":{"arxiv_id":"1910.03771","doi":"10.48550/arxiv.1910.03771","metadata_source":"pith","pith_arxiv_id":"1910.03771","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"HuggingFace's Transformers: State-of-the-art Natural Language Processing","venue":"cs.CL","work_id":"9d86da8d-01d3-41af-a0d2-ee14897927a9","year":2019},"citing_paper":{"arxiv_id":"2605.00831","last_updated":"2026-03-26T13:27:57Z","snapshot_observed_at":"2026-08-02T06:03:58.648365Z","submitted_at":"2026-03-26T13:27:57Z","title":"GhostServe: A Lightweight Checkpointing System in the Shadow for Fault-Tolerant LLM Serving","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-05-15T00:37:08.671539Z"},"links":{"cited_paper":"/paper/1910.03771","citing_paper":"/paper/2605.00831"},"observation_digest":"sha256:47f12f3dc656e1933d61fd1f13b11fcd5693f9bc47917d7fefc761dc81aad019","observation_id":"e9d8b662-e73d-445e-86ed-15b90c7c7276","resolution":{"observed_at":"2026-05-15T00:38:23.127082Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.05920","last_updated":"2024-09-25T05:57:51Z","snapshot_observed_at":"2026-07-06T15:25:24.491136Z","submitted_at":"2023-05-10T06:17:50Z","title":"Fast Distributed Inference Serving for Large Language Models","version":3},"cited_work":{"arxiv_id":"2305.05920","doi":null,"metadata_source":"pith","pith_arxiv_id":"2305.05920","snapshot_observed_at":"2026-07-04T08:39:41.965931Z","title":"Fast Distributed Inference Serving for Large Language Models","venue":"cs.LG","work_id":"30a55970-0187-4540-943d-f310a5414413","year":2023},"citing_paper":{"arxiv_id":"2605.00831","last_updated":"2026-03-26T13:27:57Z","snapshot_observed_at":"2026-08-02T06:03:58.648365Z","submitted_at":"2026-03-26T13:27:57Z","title":"GhostServe: A Lightweight Checkpointing System in the Shadow for Fault-Tolerant LLM Serving","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-05-15T00:37:08.671539Z"},"links":{"cited_paper":"/paper/2305.05920","citing_paper":"/paper/2605.00831"},"observation_digest":"sha256:3af657ff3338fe5e25fa1475fa04a64f96f9df6c484f1a126708687166acce37","observation_id":"1da15145-5a79-40c6-a3ca-a2837908f483","resolution":{"observed_at":"2026-05-17T11:23:46.844514Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.01783","last_updated":"2025-04-21T03:40:10Z","snapshot_observed_at":"2026-08-02T01:51:18.711668Z","submitted_at":"2024-11-04T04:15:36Z","title":"Context Parallelism for Scalable Million-Token Inference","version":3},"cited_work":{"arxiv_id":"2411.01783","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2411.01783","snapshot_observed_at":"2026-07-03T07:27:44.457158Z","title":"Context parallelism for scalable million-token inference","venue":null,"work_id":"12f1b1c9-af42-49fe-be5b-3b22809ea3b4","year":2024},"citing_paper":{"arxiv_id":"2605.00831","last_updated":"2026-03-26T13:27:57Z","snapshot_observed_at":"2026-08-02T06:03:58.648365Z","submitted_at":"2026-03-26T13:27:57Z","title":"GhostServe: A Lightweight Checkpointing System in the Shadow for Fault-Tolerant LLM Serving","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-05-15T00:37:08.671539Z"},"links":{"cited_paper":"/paper/2411.01783","citing_paper":"/paper/2605.00831"},"observation_digest":"sha256:c52eced68d464f9e02d5a450416b7ae829290a2952943e4cf6bd143a2fdc8740","observation_id":"5af68555-97f5-4986-ad77-b6ede239c494","resolution":{"observed_at":"2026-05-15T00:38:23.124324Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.01005","last_updated":"2025-04-21T20:10:11Z","snapshot_observed_at":"2026-07-06T20:15:36.280948Z","submitted_at":"2025-01-02T02:02:20Z","title":"FlashInfer: Efficient and Customizable Attention Engine for LLM Inference Serving","version":2},"cited_work":{"arxiv_id":"2501.01005","doi":"10.48550/arxiv.2501.01005","metadata_source":"pith","pith_arxiv_id":"2501.01005","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"FlashInfer: Efficient and Customizable Attention Engine for LLM Inference Serving","venue":"cs.DC","work_id":"a153885b-2460-4177-9053-8d0011adfcb9","year":2025},"citing_paper":{"arxiv_id":"2605.00831","last_updated":"2026-03-26T13:27:57Z","snapshot_observed_at":"2026-08-02T06:03:58.648365Z","submitted_at":"2026-03-26T13:27:57Z","title":"GhostServe: A Lightweight Checkpointing System in the Shadow for Fault-Tolerant LLM Serving","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-05-15T00:37:08.671539Z"},"links":{"cited_paper":"/paper/2501.01005","citing_paper":"/paper/2605.00831"},"observation_digest":"sha256:6a333a047ac829086c58272633779740ecd2c82f1aceb10f5c9c18b9a8fd8512","observation_id":"bffeda50-3fa6-4530-b5be-a9295d55c3e5","resolution":{"observed_at":"2026-05-16T13:26:34.840430Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.17312","last_updated":"2024-03-26T01:46:34Z","snapshot_observed_at":"2026-07-31T15:33:16.160466Z","submitted_at":"2024-03-26T01:46:34Z","title":"ALISA: Accelerating Large Language Model Inference via Sparsity-Aware KV Caching","version":1},"cited_work":{"arxiv_id":"2403.17312","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2403.17312","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Merino: Entropy-driven design for generative language models on iot devices","venue":null,"work_id":"050a9645-22c8-4e90-9bb0-8c7a3689caed","year":null},"citing_paper":{"arxiv_id":"2605.00831","last_updated":"2026-03-26T13:27:57Z","snapshot_observed_at":"2026-08-02T06:03:58.648365Z","submitted_at":"2026-03-26T13:27:57Z","title":"GhostServe: A Lightweight Checkpointing System in the Shadow for Fault-Tolerant LLM Serving","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-05-15T00:37:08.671539Z"},"links":{"cited_paper":"/paper/2403.17312","citing_paper":"/paper/2605.00831"},"observation_digest":"sha256:a14584bb9119515eb4f25831a4b07258c9e33fc9b31a2b8540a6b58a2c6c170c","observation_id":"6f8048e6-cdaa-4392-ac5a-1e1291b86f2a","resolution":{"observed_at":"2026-05-15T00:38:23.111335Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.15486","last_updated":"2025-09-03T02:11:32Z","snapshot_observed_at":"2026-07-06T18:35:08.874127Z","submitted_at":"2024-06-17T11:05:15Z","title":"SampleAttention: Near-Lossless Acceleration of Long Context LLM Inference with Adaptive Structured Sparse Attention","version":3},"cited_work":{"arxiv_id":"2406.15486","doi":"10.48550/arxiv.2406.15486","metadata_source":"arxiv_reference","pith_arxiv_id":"2406.15486","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Sam- pleattention: Near-lossless acceleration of long context llm inference with adaptive structured sparse attention","venue":"arXiv (Cornell University)","work_id":"c91a5e9e-183b-4786-a2c8-7ccef03ac295","year":2025},"citing_paper":{"arxiv_id":"2605.00831","last_updated":"2026-03-26T13:27:57Z","snapshot_observed_at":"2026-08-02T06:03:58.648365Z","submitted_at":"2026-03-26T13:27:57Z","title":"GhostServe: A Lightweight Checkpointing System in the Shadow for Fault-Tolerant LLM Serving","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-05-15T00:37:08.671539Z"},"links":{"cited_paper":"/paper/2406.15486","citing_paper":"/paper/2605.00831"},"observation_digest":"sha256:0e34e2d9d2826fa8eaaa94b00f12c81324720b85c636fdff08dabed0a8ae0e01","observation_id":"fd98a068-3906-4903-9d7c-a58fd7ffb68b","resolution":{"observed_at":"2026-05-15T00:38:23.097181Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2605.00831","last_updated":"2026-03-26T13:27:57Z","latest_version":1,"primary_category":"cs.DC","snapshot_observed_at":"2026-08-02T06:03:58.648365Z","submitted_at":"2026-03-26T13:27:57Z","title":"GhostServe: A Lightweight Checkpointing System in the Shadow for Fault-Tolerant LLM Serving"},"reference_resolution":{"displayed":24,"state_counts":{"malformed_identifier":0,"metadata_mismatch":4,"parse_uncertain":0,"unresolved":0,"verified_exact":15,"verified_fuzzy":5},"total_outbound_references":24},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"thesis":"As of 5 August 2026, this Paper Citation Record lists 24 of 24 outbound references and 0 inbound Pith citation observations for arXiv:2605.00831."}