{"as_of":"2026-08-13T02:22:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:9fec323b1e31a783dc09797ccfb8610ff71e634d989dc0c7ec1d0616071f6316","coverage":[{"denominator":37,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":37,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-12T14:07:52.471138Z","state":"measured"},{"denominator":38,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":38,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-12T06:34:41.77262+00:00","state":"measured"},{"denominator":1,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":1,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-10T20:34:34.676288Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"pith","source_observed_at":"2026-08-10T20:34:35.675801Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2411.15664","last_updated":"2024-11-23T22:19:37Z","snapshot_observed_at":"2026-08-12T21:41:08.350212Z","submitted_at":"2024-11-23T22:19:37Z","title":"Enabling Efficient Serverless Inference Serving for LLM (Large Language Model) in the Cloud","version":1},"cited_work":{"arxiv_id":"2411.15664","doi":null,"metadata_source":"pith","pith_arxiv_id":"2411.15664","snapshot_observed_at":"2026-08-10T20:34:35.675801Z","title":"Enabling Efficient Serverless Inference Serving for LLM (Large Language Model) in the Cloud","venue":"cs.DC","work_id":"85663b51-fd5e-4061-90d6-dbdbfab74213","year":2024},"citing_paper":{"arxiv_id":"2501.08262","last_updated":"2025-01-14T17:21:16Z","snapshot_observed_at":"2026-08-12T12:12:48.696575Z","submitted_at":"2025-01-14T17:21:16Z","title":"Addressing the sustainable AI trilemma: a case study on LLM agents and RAG","version":1},"reference_index":2024,"source":"pdf_text","source_observed_at":"2026-08-10T20:34:34.676288Z"},"links":{"cited_paper":"/paper/2411.15664","citing_paper":"/paper/2501.08262"},"observation_digest":"sha256:23d6f8a61b40e4c92c143d4824c805d22b3af3eef4f30cde279ff875edbf6a44","observation_id":"36daebe2-8f27-449a-814d-24b40e8aae63","resolution":{"observed_at":"2026-08-10T20:34:35.680096Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2411.15664/citation-record","integrity":"/paper/2411.15664/integrity","json":"/paper/2411.15664/citation-record.json","paper":"/paper/2411.15664"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T14:07:53.202718Z","title":"slideshare","venue":null,"work_id":"d06de8ff-6561-4e7c-a2a7-7ae51b04ed1a","year":2017},"citing_paper":{"arxiv_id":"2411.15664","last_updated":"2024-11-23T22:19:37Z","snapshot_observed_at":"2026-08-12T21:41:08.350212Z","submitted_at":"2024-11-23T22:19:37Z","title":"Enabling Efficient Serverless Inference Serving for LLM (Large Language Model) in the Cloud","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-12T14:07:52.342326Z"},"links":{"citing_paper":"/paper/2411.15664"},"observation_digest":"sha256:4a168aec977edb32e52df18ec4d72a4bbdaed58a58e669df277ca3713fa21e61","observation_id":"364f3b2d-7051-41f6-93c8-dedeace87310","resolution":{"observed_at":"2026-08-12T14:07:53.205858Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T14:07:53.192069Z","title":null,"venue":null,"work_id":"b4812003-e656-4c3d-b95b-5e679edab27b","year":null},"citing_paper":{"arxiv_id":"2411.15664","last_updated":"2024-11-23T22:19:37Z","snapshot_observed_at":"2026-08-12T21:41:08.350212Z","submitted_at":"2024-11-23T22:19:37Z","title":"Enabling Efficient Serverless Inference Serving for LLM (Large Language Model) in the Cloud","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-12T14:07:52.346076Z"},"links":{"citing_paper":"/paper/2411.15664"},"observation_digest":"sha256:c756c1be0320df450cb9187a60ed1bb19c3faf4924fe74bcf3240cafaa998bc9","observation_id":"5df39438-f0dd-40b9-b554-337378de84b6","resolution":{"observed_at":"2026-08-12T14:07:53.196280Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T14:07:53.181923Z","title":null,"venue":null,"work_id":"06d3cb0e-d614-4b23-a7be-03f5cba4ed2d","year":null},"citing_paper":{"arxiv_id":"2411.15664","last_updated":"2024-11-23T22:19:37Z","snapshot_observed_at":"2026-08-12T21:41:08.350212Z","submitted_at":"2024-11-23T22:19:37Z","title":"Enabling Efficient Serverless Inference Serving for LLM (Large Language Model) in the Cloud","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-12T14:07:52.349499Z"},"links":{"citing_paper":"/paper/2411.15664"},"observation_digest":"sha256:af804ebbfd261b208b8f0162051fd259816867375f89a4345353a2a45ecc5340","observation_id":"1023118d-9954-4aca-9d91-080b1315d27e","resolution":{"observed_at":"2026-08-12T14:07:53.185421Z","resolver_source":"raw_fallback","status":"parse_uncertain"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T14:07:53.171066Z","title":"com/ jeremydaly/ lambda-warmer","venue":null,"work_id":"40d2fecf-b76f-4407-ad45-fa5c043797d8","year":null},"citing_paper":{"arxiv_id":"2411.15664","last_updated":"2024-11-23T22:19:37Z","snapshot_observed_at":"2026-08-12T21:41:08.350212Z","submitted_at":"2024-11-23T22:19:37Z","title":"Enabling Efficient Serverless Inference Serving for LLM (Large Language Model) in the Cloud","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-12T14:07:52.353297Z"},"links":{"citing_paper":"/paper/2411.15664"},"observation_digest":"sha256:6729c8e5833cf0776b61d4606564d9c07787362d0e75e985a0978eff01af6f7c","observation_id":"feb893f8-64be-4fa3-b183-850deaffcb9b","resolution":{"observed_at":"2026-08-12T14:07:53.175555Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T14:07:53.161090Z","title":"com/ google-cloud/ 3-solutions-to-mitigate-the-cold-starts-on-cloud-run-8c60f0ae7894","venue":null,"work_id":"f391a95b-bcfa-4adb-87c4-073064e0f270","year":null},"citing_paper":{"arxiv_id":"2411.15664","last_updated":"2024-11-23T22:19:37Z","snapshot_observed_at":"2026-08-12T21:41:08.350212Z","submitted_at":"2024-11-23T22:19:37Z","title":"Enabling Efficient Serverless Inference Serving for LLM (Large Language Model) in the Cloud","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-12T14:07:52.356999Z"},"links":{"citing_paper":"/paper/2411.15664"},"observation_digest":"sha256:164d988d684f0488cc15f607d892f49fb745da281ec9fe87e1a8b0906b9c80f0","observation_id":"d3177d5b-5d1d-4900-a5f4-aa86aa6436e9","resolution":{"observed_at":"2026-08-12T14:07:53.164567Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T14:07:53.151642Z","title":null,"venue":null,"work_id":"baf5967d-b9b2-4034-b2f4-fce4c57433f6","year":null},"citing_paper":{"arxiv_id":"2411.15664","last_updated":"2024-11-23T22:19:37Z","snapshot_observed_at":"2026-08-12T21:41:08.350212Z","submitted_at":"2024-11-23T22:19:37Z","title":"Enabling Efficient Serverless Inference Serving for LLM (Large Language Model) in the Cloud","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-12T14:07:52.360480Z"},"links":{"citing_paper":"/paper/2411.15664"},"observation_digest":"sha256:70fe6aa36a46a74f21febc159000b0363438a1ebaba26311fee2836d22a80fa8","observation_id":"2f795c3d-195c-42f5-8547-405954d828c8","resolution":{"observed_at":"2026-08-12T14:07:53.154936Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T14:07:53.143081Z","title":null,"venue":null,"work_id":"f40c6d88-2783-4d67-8a2d-239fc40aca91","year":null},"citing_paper":{"arxiv_id":"2411.15664","last_updated":"2024-11-23T22:19:37Z","snapshot_observed_at":"2026-08-12T21:41:08.350212Z","submitted_at":"2024-11-23T22:19:37Z","title":"Enabling Efficient Serverless Inference Serving for LLM (Large Language Model) in the Cloud","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-12T14:07:52.363413Z"},"links":{"citing_paper":"/paper/2411.15664"},"observation_digest":"sha256:9723d9195d131b80263dfb4e3debde6768536eaa5a5befb7f5203dc30e69da5f","observation_id":"3d03005d-8e8d-46a6-8ae3-291f4f5dbaba","resolution":{"observed_at":"2026-08-12T14:07:53.145956Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T14:07:53.133732Z","title":"microsoft","venue":null,"work_id":"7f7c28d2-4115-4c9f-96d7-8675494c0b75","year":null},"citing_paper":{"arxiv_id":"2411.15664","last_updated":"2024-11-23T22:19:37Z","snapshot_observed_at":"2026-08-12T21:41:08.350212Z","submitted_at":"2024-11-23T22:19:37Z","title":"Enabling Efficient Serverless Inference Serving for LLM (Large Language Model) in the Cloud","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-12T14:07:52.366346Z"},"links":{"citing_paper":"/paper/2411.15664"},"observation_digest":"sha256:3b70b8d9392f574da6503534ad0794eeb4c92b79450a3b0766bbef034dc009f8","observation_id":"202f8e13-e3c3-4812-bcdc-c0c01875a679","resolution":{"observed_at":"2026-08-12T14:07:53.137112Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T14:07:53.123556Z","title":null,"venue":null,"work_id":"909adf15-1637-4748-ad59-16b4f8c42d4c","year":null},"citing_paper":{"arxiv_id":"2411.15664","last_updated":"2024-11-23T22:19:37Z","snapshot_observed_at":"2026-08-12T21:41:08.350212Z","submitted_at":"2024-11-23T22:19:37Z","title":"Enabling Efficient Serverless Inference Serving for LLM (Large Language Model) in the Cloud","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-12T14:07:52.368997Z"},"links":{"citing_paper":"/paper/2411.15664"},"observation_digest":"sha256:1244330e40612b5d14cd0493922b9a6ddbb86813f5ec149cb41e262f7460660b","observation_id":"de4aaac9-7e90-4eba-a5dc-089d7fca0403","resolution":{"observed_at":"2026-08-12T14:07:53.127094Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T14:07:53.114066Z","title":"https: // www","venue":null,"work_id":"8f7c4cf4-a072-4cfe-9b4c-ceb148f18746","year":null},"citing_paper":{"arxiv_id":"2411.15664","last_updated":"2024-11-23T22:19:37Z","snapshot_observed_at":"2026-08-12T21:41:08.350212Z","submitted_at":"2024-11-23T22:19:37Z","title":"Enabling Efficient Serverless Inference Serving for LLM (Large Language Model) in the Cloud","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-12T14:07:52.371989Z"},"links":{"citing_paper":"/paper/2411.15664"},"observation_digest":"sha256:8b803e2fbf6104390661f517b6073a56287f50f79bf1d5909891eeb7a265e4b6","observation_id":"2fcba9bd-9c35-43a4-a845-12c0b7fa35f5","resolution":{"observed_at":"2026-08-12T14:07:53.117528Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T14:07:53.104296Z","title":"SedAI ( https: // www","venue":null,"work_id":"9c7c06ed-41e4-436b-b8c9-e77e6130a8b9","year":null},"citing_paper":{"arxiv_id":"2411.15664","last_updated":"2024-11-23T22:19:37Z","snapshot_observed_at":"2026-08-12T21:41:08.350212Z","submitted_at":"2024-11-23T22:19:37Z","title":"Enabling Efficient Serverless Inference Serving for LLM (Large Language Model) in the Cloud","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-12T14:07:52.375096Z"},"links":{"citing_paper":"/paper/2411.15664"},"observation_digest":"sha256:67775de60da1e5aa10e8208b8c0a29c49688561bc2e8aaa6e3ce1d845fdf9c65","observation_id":"82eec83a-4d1c-40dd-b542-37ad77f34f47","resolution":{"observed_at":"2026-08-12T14:07:53.107679Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T14:07:53.094314Z","title":"Snowflake ( https: // www","venue":null,"work_id":"15b525ab-50b9-4a74-b18c-5df9536a331e","year":null},"citing_paper":{"arxiv_id":"2411.15664","last_updated":"2024-11-23T22:19:37Z","snapshot_observed_at":"2026-08-12T21:41:08.350212Z","submitted_at":"2024-11-23T22:19:37Z","title":"Enabling Efficient Serverless Inference Serving for LLM (Large Language Model) in the Cloud","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-12T14:07:52.379176Z"},"links":{"citing_paper":"/paper/2411.15664"},"observation_digest":"sha256:0f8c02f0daf65128d4f98ddd309ddbfe41aeec6ce0b14a708ff4e11e9d871bd1","observation_id":"a4c88b13-02c7-44e8-a963-1a4f7d05878c","resolution":{"observed_at":"2026-08-12T14:07:53.097763Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T14:07:53.083961Z","title":"( https: // en","venue":null,"work_id":"183b65c4-b001-481f-b5c7-45c2ce11c37a","year":null},"citing_paper":{"arxiv_id":"2411.15664","last_updated":"2024-11-23T22:19:37Z","snapshot_observed_at":"2026-08-12T21:41:08.350212Z","submitted_at":"2024-11-23T22:19:37Z","title":"Enabling Efficient Serverless Inference Serving for LLM (Large Language Model) in the Cloud","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-12T14:07:52.382791Z"},"links":{"citing_paper":"/paper/2411.15664"},"observation_digest":"sha256:4a672041592c5c1e38e686cb446b457a3e9d845fbeaa64d9f7e1f030a91e74ee","observation_id":"4324c903-1289-4ab5-a36b-8b9f2f5b0cde","resolution":{"observed_at":"2026-08-12T14:07:53.088216Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.05920","last_updated":"2024-09-25T05:57:51Z","snapshot_observed_at":"2026-08-09T14:08:47.340904Z","submitted_at":"2023-05-10T06:17:50Z","title":"Fast Distributed Inference Serving for Large Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.05920","snapshot_observed_at":"2026-08-12T14:07:52.386091Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2411.15664","last_updated":"2024-11-23T22:19:37Z","snapshot_observed_at":"2026-08-12T21:41:08.350212Z","submitted_at":"2024-11-23T22:19:37Z","title":"Enabling Efficient Serverless Inference Serving for LLM (Large Language Model) in the Cloud","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-12T14:07:52.386091Z"},"links":{"cited_paper":"/paper/2305.05920","citing_paper":"/paper/2411.15664"},"observation_digest":"sha256:e2964303279e4e4811a687368b9ccd93014df83736010d89fa62aa3bc4a7b271","observation_id":"45d9e220-c714-4936-8a6c-48e75dbbadc6","resolution":{"observed_at":"2026-08-12T14:07:52.386091Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T14:07:53.074773Z","title":null,"venue":null,"work_id":"decc58aa-b22a-4217-bd06-e412fa908837","year":null},"citing_paper":{"arxiv_id":"2411.15664","last_updated":"2024-11-23T22:19:37Z","snapshot_observed_at":"2026-08-12T21:41:08.350212Z","submitted_at":"2024-11-23T22:19:37Z","title":"Enabling Efficient Serverless Inference Serving for LLM (Large Language Model) in the Cloud","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-12T14:07:52.390158Z"},"links":{"citing_paper":"/paper/2411.15664"},"observation_digest":"sha256:471dbe2c38a7f7cfa8090629b33902e4904464ef9182de774db67ba49514a3d9","observation_id":"de6462a3-8c62-4370-b450-504444aa0743","resolution":{"observed_at":"2026-08-12T14:07:53.077833Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T14:07:52.398677Z","title":"Catalyzer: Sub-millisecond startup for serverless computing with initialization-less booting","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2411.15664","last_updated":"2024-11-23T22:19:37Z","snapshot_observed_at":"2026-08-12T21:41:08.350212Z","submitted_at":"2024-11-23T22:19:37Z","title":"Enabling Efficient Serverless Inference Serving for LLM (Large Language Model) in the Cloud","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-12T14:07:52.398677Z"},"links":{"citing_paper":"/paper/2411.15664"},"observation_digest":"sha256:44f514a66585c88a728d9ffc15f98a1d01108b4eb947d58470e68313be451ad7","observation_id":"68f846e0-6afe-439f-9d34-8d3fd4386f23","resolution":{"observed_at":"2026-08-12T14:07:52.398677Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T14:07:53.056065Z","title":null,"venue":null,"work_id":"82e5e216-739d-4fd7-93ed-e470860f459f","year":2018},"citing_paper":{"arxiv_id":"2411.15664","last_updated":"2024-11-23T22:19:37Z","snapshot_observed_at":"2026-08-12T21:41:08.350212Z","submitted_at":"2024-11-23T22:19:37Z","title":"Enabling Efficient Serverless Inference Serving for LLM (Large Language Model) in the Cloud","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-12T14:07:52.402455Z"},"links":{"citing_paper":"/paper/2411.15664"},"observation_digest":"sha256:89dac1c38f33995962167751503b7de7145d7a4d4450bcac0d02b8e85ff2a798","observation_id":"4e46f8db-aba4-4517-846a-6e2e7015b7ef","resolution":{"observed_at":"2026-08-12T14:07:53.059267Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T14:07:52.406118Z","title":"Centralized core-granular scheduling for serverless functions","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2411.15664","last_updated":"2024-11-23T22:19:37Z","snapshot_observed_at":"2026-08-12T21:41:08.350212Z","submitted_at":"2024-11-23T22:19:37Z","title":"Enabling Efficient Serverless Inference Serving for LLM (Large Language Model) in the Cloud","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-12T14:07:52.406118Z"},"links":{"citing_paper":"/paper/2411.15664"},"observation_digest":"sha256:cf9021d1035ea5923c9342b2bfa3271f49ea2f4644a75908765b83e36493eb6a","observation_id":"12f96f92-095c-409e-95ce-7c044a2ed2ec","resolution":{"observed_at":"2026-08-12T14:07:52.406118Z","resolver_source":null,"status":"malformed_identifier"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T14:07:53.065583Z","title":null,"venue":null,"work_id":"b01e7b51-acb0-406a-a835-507d97257b69","year":2019},"citing_paper":{"arxiv_id":"2411.15664","last_updated":"2024-11-23T22:19:37Z","snapshot_observed_at":"2026-08-12T21:41:08.350212Z","submitted_at":"2024-11-23T22:19:37Z","title":"Enabling Efficient Serverless Inference Serving for LLM (Large Language Model) in the Cloud","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-12T14:07:52.394148Z"},"links":{"citing_paper":"/paper/2411.15664"},"observation_digest":"sha256:330096eca014433171743eb1eb16f5cb0ad38d14812d74d44602b4be1e058941","observation_id":"0f092b08-7059-4c5c-ae16-0930f2ba1aff","resolution":{"observed_at":"2026-08-12T14:07:53.069167Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T14:07:53.046095Z","title":"Faaslight: General application-level cold-start latency optimization for function-as-a-service in serverless comput- ing","venue":null,"work_id":"d6955314-6f2b-4e0d-ab0a-07725747a928","year":2023},"citing_paper":{"arxiv_id":"2411.15664","last_updated":"2024-11-23T22:19:37Z","snapshot_observed_at":"2026-08-12T21:41:08.350212Z","submitted_at":"2024-11-23T22:19:37Z","title":"Enabling Efficient Serverless Inference Serving for LLM (Large Language Model) in the Cloud","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-12T14:07:52.409303Z"},"links":{"citing_paper":"/paper/2411.15664"},"observation_digest":"sha256:633de8095e84bf350cf8030adba1ea9e54e8b8542a1eff895d488f502303a059","observation_id":"db06b321-02d0-434d-8d5e-5a639575bc15","resolution":{"observed_at":"2026-08-12T14:07:53.049715Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T14:07:53.035802Z","title":"Rapid task provision- ing with Serverless-Optimized containers","venue":null,"work_id":"55043c20-d7f9-4873-845a-8da39b54feef","year":2018},"citing_paper":{"arxiv_id":"2411.15664","last_updated":"2024-11-23T22:19:37Z","snapshot_observed_at":"2026-08-12T21:41:08.350212Z","submitted_at":"2024-11-23T22:19:37Z","title":"Enabling Efficient Serverless Inference Serving for LLM (Large Language Model) in the Cloud","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-12T14:07:52.413546Z"},"links":{"citing_paper":"/paper/2411.15664"},"observation_digest":"sha256:5562d299055aa91ba6254530fbf0eb8dc8d4f0a23603057af47e6a238507ad14","observation_id":"5a4c2f0e-8182-4dd1-8406-34b232294aaf","resolution":{"observed_at":"2026-08-12T14:07:53.039443Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T14:07:53.025598Z","title":"Ghobaei-Arani","venue":null,"work_id":"84a8829e-22b7-42d3-9967-3d84e6b0965d","year":2024},"citing_paper":{"arxiv_id":"2411.15664","last_updated":"2024-11-23T22:19:37Z","snapshot_observed_at":"2026-08-12T21:41:08.350212Z","submitted_at":"2024-11-23T22:19:37Z","title":"Enabling Efficient Serverless Inference Serving for LLM (Large Language Model) in the Cloud","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-12T14:07:52.416575Z"},"links":{"citing_paper":"/paper/2411.15664"},"observation_digest":"sha256:aff7d7c5b607f01653e8f0d8be3fb7fcadf6d68ce413688d819eeeced87b57f3","observation_id":"b4ba1b40-1a9f-4c4a-b1e0-f48fe07ae907","resolution":{"observed_at":"2026-08-12T14:07:53.029020Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T14:07:53.016253Z","title":"Cold start latency in serverless com- puting: A systematic review, taxonomy, and future directions","venue":null,"work_id":"81968d78-5070-4721-b6f3-66a6670c9a5b","year":null},"citing_paper":{"arxiv_id":"2411.15664","last_updated":"2024-11-23T22:19:37Z","snapshot_observed_at":"2026-08-12T21:41:08.350212Z","submitted_at":"2024-11-23T22:19:37Z","title":"Enabling Efficient Serverless Inference Serving for LLM (Large Language Model) in the Cloud","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-12T14:07:52.419557Z"},"links":{"citing_paper":"/paper/2411.15664"},"observation_digest":"sha256:95994667a3abb1597386258b11f0cca1497326cf5d5a324006f0f19881e31f20","observation_id":"79557d58-7223-4ac3-8d6a-aa67466ddae0","resolution":{"observed_at":"2026-08-12T14:07:53.019552Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T14:07:53.005922Z","title":"What is serverless computing? IBM ( https: // www","venue":null,"work_id":"6d31a2d5-0567-4aad-ba56-3f06d6f9eab3","year":null},"citing_paper":{"arxiv_id":"2411.15664","last_updated":"2024-11-23T22:19:37Z","snapshot_observed_at":"2026-08-12T21:41:08.350212Z","submitted_at":"2024-11-23T22:19:37Z","title":"Enabling Efficient Serverless Inference Serving for LLM (Large Language Model) in the Cloud","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-12T14:07:52.422766Z"},"links":{"citing_paper":"/paper/2411.15664"},"observation_digest":"sha256:f2d49818abb76c91eadb15bbd772df79536f2a44b89e185dae9d5771a743943b","observation_id":"b542313f-fb5d-4c01-84dd-f9165069a116","resolution":{"observed_at":"2026-08-12T14:07:53.009677Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1903.12221","last_updated":"2019-03-28T18:55:30Z","snapshot_observed_at":"2026-07-06T07:42:24.629331Z","submitted_at":"2019-03-28T18:55:30Z","title":"Mitigating Cold Starts in Serverless Platforms: A Pool-Based Approach","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1903.12221","snapshot_observed_at":"2026-08-12T14:07:52.425584Z","title":"Mitigat- ing cold starts in serverless platforms: A pool-based approach https://arxiv.org/abs/ 1903.12221, 2019","venue":null,"work_id":null,"year":1903},"citing_paper":{"arxiv_id":"2411.15664","last_updated":"2024-11-23T22:19:37Z","snapshot_observed_at":"2026-08-12T21:41:08.350212Z","submitted_at":"2024-11-23T22:19:37Z","title":"Enabling Efficient Serverless Inference Serving for LLM (Large Language Model) in the Cloud","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-12T14:07:52.425584Z"},"links":{"cited_paper":"/paper/1903.12221","citing_paper":"/paper/2411.15664"},"observation_digest":"sha256:4c672e7bbb9c639d41008c8b369d41586d6b0adc0cedcd372017f2ead2e888cd","observation_id":"33d95291-1641-4918-a0e1-733e3aaf5e5f","resolution":{"observed_at":"2026-08-12T14:07:52.425584Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T14:07:52.994997Z","title":null,"venue":null,"work_id":"d651aa6d-9a27-4e8a-9cbc-aac003b843cd","year":2017},"citing_paper":{"arxiv_id":"2411.15664","last_updated":"2024-11-23T22:19:37Z","snapshot_observed_at":"2026-08-12T21:41:08.350212Z","submitted_at":"2024-11-23T22:19:37Z","title":"Enabling Efficient Serverless Inference Serving for LLM (Large Language Model) in the Cloud","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-12T14:07:52.429465Z"},"links":{"citing_paper":"/paper/2411.15664"},"observation_digest":"sha256:9b118e1e013c6f9ba85680d0ceb2987b91f6a27044592abbf685bd5655ecdfec","observation_id":"11398b45-2b56-40d3-9991-a9412e7f1d4f","resolution":{"observed_at":"2026-08-12T14:07:52.998521Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T14:07:52.983881Z","title":"Persson and W","venue":null,"work_id":"032c0c82-39b9-431e-b6ba-2523713213a4","year":null},"citing_paper":{"arxiv_id":"2411.15664","last_updated":"2024-11-23T22:19:37Z","snapshot_observed_at":"2026-08-12T21:41:08.350212Z","submitted_at":"2024-11-23T22:19:37Z","title":"Enabling Efficient Serverless Inference Serving for LLM (Large Language Model) in the Cloud","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-12T14:07:52.433660Z"},"links":{"citing_paper":"/paper/2411.15664"},"observation_digest":"sha256:90ffd262db520845b204efbd2b71696e68f5aa4c785f590f656d1da390093c52","observation_id":"0e5868b2-1ac2-456a-bfa0-1064a03ebf44","resolution":{"observed_at":"2026-08-12T14:07:52.987551Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T14:07:52.437985Z","title":null,"venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2411.15664","last_updated":"2024-11-23T22:19:37Z","snapshot_observed_at":"2026-08-12T21:41:08.350212Z","submitted_at":"2024-11-23T22:19:37Z","title":"Enabling Efficient Serverless Inference Serving for LLM (Large Language Model) in the Cloud","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-12T14:07:52.437985Z"},"links":{"citing_paper":"/paper/2411.15664"},"observation_digest":"sha256:6b9ff22d96cca6ad4bf3d47ca5d87d95e05509d3701e31f3d19af85a0988ce63","observation_id":"b4ad41ec-044c-48f0-96e5-3288dc7dee95","resolution":{"observed_at":"2026-08-12T14:07:52.437985Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T14:07:52.971603Z","title":"Pietzuch https: //api.semanticscholar.org/CorpusID: 11 51997872","venue":null,"work_id":"e1acfd42-61f7-44fe-93d3-642e295fc4e8","year":2018},"citing_paper":{"arxiv_id":"2411.15664","last_updated":"2024-11-23T22:19:37Z","snapshot_observed_at":"2026-08-12T21:41:08.350212Z","submitted_at":"2024-11-23T22:19:37Z","title":"Enabling Efficient Serverless Inference Serving for LLM (Large Language Model) in the Cloud","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-12T14:07:52.441604Z"},"links":{"citing_paper":"/paper/2411.15664"},"observation_digest":"sha256:93f1a0332af891c49365906535c0cb37a2a44719dfd358585a2f8503a44afd2f","observation_id":"14f82162-c99a-4d65-b6f0-640b1c60b349","resolution":{"observed_at":"2026-08-12T14:07:52.975524Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T14:07:52.959966Z","title":null,"venue":null,"work_id":"a21c7f2c-acc1-44e2-a926-abefafcbe5b8","year":2020},"citing_paper":{"arxiv_id":"2411.15664","last_updated":"2024-11-23T22:19:37Z","snapshot_observed_at":"2026-08-12T21:41:08.350212Z","submitted_at":"2024-11-23T22:19:37Z","title":"Enabling Efficient Serverless Inference Serving for LLM (Large Language Model) in the Cloud","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-12T14:07:52.445502Z"},"links":{"citing_paper":"/paper/2411.15664"},"observation_digest":"sha256:ecd89eb204aa58374a813214f00b14887cf4113caaf943e00f0665a203f7968e","observation_id":"f63de849-60e9-4273-bc54-25fbc1fda6b5","resolution":{"observed_at":"2026-08-12T14:07:52.964010Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2206.12275","last_updated":"2022-12-16T13:22:26Z","snapshot_observed_at":"2026-08-12T18:12:54.708824Z","submitted_at":"2022-06-24T13:25:55Z","title":"Rise of the Planet of Serverless Computing: A Systematic Review","version":5},"cited_work":{"arxiv_id":"2206.12275","doi":null,"metadata_source":"pith","pith_arxiv_id":"2206.12275","snapshot_observed_at":"2026-08-12T14:07:52.668814Z","title":"Rise of the Planet of Serverless Computing: A Systematic Review","venue":"cs.SE","work_id":"6ddcb9d4-5d9e-4660-8b09-b6fb96148146","year":2022},"citing_paper":{"arxiv_id":"2411.15664","last_updated":"2024-11-23T22:19:37Z","snapshot_observed_at":"2026-08-12T21:41:08.350212Z","submitted_at":"2024-11-23T22:19:37Z","title":"Enabling Efficient Serverless Inference Serving for LLM (Large Language Model) in the Cloud","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-12T14:07:52.448829Z"},"links":{"cited_paper":"/paper/2206.12275","citing_paper":"/paper/2411.15664"},"observation_digest":"sha256:745be9a04b9ff0a8447e1f2f771c7a91e825ca05e60c768035c1c26bfd53ff50","observation_id":"eb9f7d2c-ddeb-4acc-95b0-dd7eff3418a4","resolution":{"observed_at":"2026-08-12T14:07:52.674384Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T14:07:52.948797Z","title":null,"venue":null,"work_id":"9a3f6de9-7295-48fd-80e8-e89b8a8164a3","year":null},"citing_paper":{"arxiv_id":"2411.15664","last_updated":"2024-11-23T22:19:37Z","snapshot_observed_at":"2026-08-12T21:41:08.350212Z","submitted_at":"2024-11-23T22:19:37Z","title":"Enabling Efficient Serverless Inference Serving for LLM (Large Language Model) in the Cloud","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-12T14:07:52.452351Z"},"links":{"citing_paper":"/paper/2411.15664"},"observation_digest":"sha256:636dbf3bbadb032e2172334d471e3e40005897c931762b49a8ffd175b680928a","observation_id":"727ca945-9b4f-450d-a8cd-e42ab8a30ace","resolution":{"observed_at":"2026-08-12T14:07:52.952132Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2303.06865","last_updated":"2023-06-12T07:48:53Z","snapshot_observed_at":"2026-08-06T08:43:40.051791Z","submitted_at":"2023-03-13T05:19:28Z","title":"FlexGen: High-Throughput Generative Inference of Large Language Models with a Single GPU","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.06865","snapshot_observed_at":"2026-08-12T14:07:52.463915Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2411.15664","last_updated":"2024-11-23T22:19:37Z","snapshot_observed_at":"2026-08-12T21:41:08.350212Z","submitted_at":"2024-11-23T22:19:37Z","title":"Enabling Efficient Serverless Inference Serving for LLM (Large Language Model) in the Cloud","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-12T14:07:52.463915Z"},"links":{"cited_paper":"/paper/2303.06865","citing_paper":"/paper/2411.15664"},"observation_digest":"sha256:53e387d5d021d879489a5c062b0dd319c4dc5af2aeeb07eb6c05cdbb26905f4d","observation_id":"3aeb15bc-302b-4493-a87b-30bbfd068181","resolution":{"observed_at":"2026-08-12T14:07:52.463915Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T14:07:52.460464Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2411.15664","last_updated":"2024-11-23T22:19:37Z","snapshot_observed_at":"2026-08-12T21:41:08.350212Z","submitted_at":"2024-11-23T22:19:37Z","title":"Enabling Efficient Serverless Inference Serving for LLM (Large Language Model) in the Cloud","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-12T14:07:52.460464Z"},"links":{"citing_paper":"/paper/2411.15664"},"observation_digest":"sha256:2f1045de4292f1b94c784c977938f43405b2d666253f2c8177fc24f4aee8963f","observation_id":"0daaa7ee-668f-4558-bc16-d7a6f5ed40da","resolution":{"observed_at":"2026-08-12T14:07:52.460464Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T14:07:52.938026Z","title":"Taming serverless cold start of cloud model inference with edge comput- ing","venue":null,"work_id":"f81e1691-ebf3-4e5c-bdaa-7b4e2790f252","year":2024},"citing_paper":{"arxiv_id":"2411.15664","last_updated":"2024-11-23T22:19:37Z","snapshot_observed_at":"2026-08-12T21:41:08.350212Z","submitted_at":"2024-11-23T22:19:37Z","title":"Enabling Efficient Serverless Inference Serving for LLM (Large Language Model) in the Cloud","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-12T14:07:52.471138Z"},"links":{"citing_paper":"/paper/2411.15664"},"observation_digest":"sha256:247052db10219ab22df0849df91f77c1c72c02f819d91a61d39231f37a409f70","observation_id":"ade5bb26-b9e9-4a20-97a5-b1e61e5ec008","resolution":{"observed_at":"2026-08-12T14:07:52.942131Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-12T06:34:41.77262+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-12T14:07:52.467684Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2411.15664","last_updated":"2024-11-23T22:19:37Z","snapshot_observed_at":"2026-08-12T21:41:08.350212Z","submitted_at":"2024-11-23T22:19:37Z","title":"Enabling Efficient Serverless Inference Serving for LLM (Large Language Model) in the Cloud","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-12T14:07:52.467684Z"},"links":{"citing_paper":"/paper/2411.15664"},"observation_digest":"sha256:28e1c24c03082d86701d7d3fdbb10cc08f0a4380d3733929146f13e4a4ebdd2d","observation_id":"11983ed9-bda9-45a0-bd46-fcb59feb7f7c","resolution":{"observed_at":"2026-08-12T14:07:52.467684Z","resolver_source":null,"status":"malformed_identifier"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.14351","last_updated":"2024-07-25T08:08:11Z","snapshot_observed_at":"2026-08-12T18:12:20.817710Z","submitted_at":"2024-01-25T17:55:07Z","title":"ServerlessLLM: Low-Latency Serverless Inference for Large Language Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.14351","snapshot_observed_at":"2026-08-12T14:07:52.456185Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2411.15664","last_updated":"2024-11-23T22:19:37Z","snapshot_observed_at":"2026-08-12T21:41:08.350212Z","submitted_at":"2024-11-23T22:19:37Z","title":"Enabling Efficient Serverless Inference Serving for LLM (Large Language Model) in the Cloud","version":1},"reference_index":2024,"source":"pdf_text","source_observed_at":"2026-08-12T14:07:52.456185Z"},"links":{"cited_paper":"/paper/2401.14351","citing_paper":"/paper/2411.15664"},"observation_digest":"sha256:f05f94762af454f66abccb42542857f3db8f386b62710ae36d6fed2a927f85a5","observation_id":"ae7ea405-8a5d-4248-bc6f-6f03e4a0382b","resolution":{"observed_at":"2026-08-12T14:07:52.456185Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2411.15664","last_updated":"2024-11-23T22:19:37Z","latest_version":1,"primary_category":"cs.DC","snapshot_observed_at":"2026-08-12T21:41:08.350212Z","submitted_at":"2024-11-23T22:19:37Z","title":"Enabling Efficient Serverless Inference Serving for LLM (Large Language Model) in the Cloud"},"reference_resolution":{"displayed":37,"state_counts":{"malformed_identifier":2,"metadata_mismatch":0,"parse_uncertain":1,"unresolved":17,"verified_exact":1,"verified_fuzzy":16},"total_outbound_references":37},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-12T06:34:41.77262+00:00","source":"crossref"},{"observed_at":"2026-08-12T06:34:36.333875+00:00","source":"retraction_watch"}],"thesis":"As of 13 August 2026, this Paper Citation Record lists 37 of 37 outbound references and 1 inbound Pith citation observation for arXiv:2411.15664."}