{"as_of":"2026-08-07T10:32:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:685d4d72f728d9947a1ce34bf3e14063aee9efbc634c9b0ff0208f4bec86366a","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":10,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":10,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-07T06:34:17.273281+00:00","state":"measured"},{"denominator":10,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":10,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T05:56:56.766216Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"pith","source_observed_at":"2026-07-10T00:56:40.976559Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2312.15234","last_updated":"2025-07-23T10:11:55Z","snapshot_observed_at":"2026-08-05T06:23:34.302204Z","submitted_at":"2023-12-23T11:57:53Z","title":"Towards Efficient Generative Large Language Model Serving: A Survey from Algorithms to Systems","version":2},"cited_work":{"arxiv_id":"2312.15234","doi":null,"metadata_source":"pith","pith_arxiv_id":"2312.15234","snapshot_observed_at":"2026-07-10T00:56:40.976559Z","title":"Towards efficient generative large language model serving: A survey from algorithms to systems","venue":"cs.LG","work_id":"60d8ed33-eba3-401c-9124-609458817b15","year":2023},"citing_paper":{"arxiv_id":"2404.13501","last_updated":"2024-04-21T01:49:46Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-04-21T01:49:46Z","title":"A Survey on the Memory Mechanism of Large Language Model based Agents","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-05-15T07:21:39.440092Z"},"links":{"cited_paper":"/paper/2312.15234","citing_paper":"/paper/2404.13501"},"observation_digest":"sha256:d3c3da59f73afdf172deb7371b2feabc09fc31f6b58232d800f196f36f3fd38f","observation_id":"6a98f694-96d1-4510-8e9c-4ea55debfff5","resolution":{"observed_at":"2026-05-15T07:21:39.758995Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.15234","last_updated":"2025-07-23T10:11:55Z","snapshot_observed_at":"2026-08-05T06:23:34.302204Z","submitted_at":"2023-12-23T11:57:53Z","title":"Towards Efficient Generative Large Language Model Serving: A Survey from Algorithms to Systems","version":2},"cited_work":{"arxiv_id":"2312.15234","doi":null,"metadata_source":"pith","pith_arxiv_id":"2312.15234","snapshot_observed_at":"2026-07-10T00:56:40.976559Z","title":"Towards efficient generative large language model serving: A survey from algorithms to systems","venue":"cs.LG","work_id":"60d8ed33-eba3-401c-9124-609458817b15","year":2023},"citing_paper":{"arxiv_id":"2404.14294","last_updated":"2024-07-19T04:47:36Z","snapshot_observed_at":"2026-07-06T18:03:47.096406Z","submitted_at":"2024-04-22T15:53:08Z","title":"A Survey on Efficient Inference for Large Language Models","version":3},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-05-15T02:39:33.007894Z"},"links":{"cited_paper":"/paper/2312.15234","citing_paper":"/paper/2404.14294"},"observation_digest":"sha256:3f7b68b7b83eb2d7b30db67137856794a0ef636606db5efc4de43936564e9cd9","observation_id":"46f675b4-31b2-43c6-8440-a01ed82e3350","resolution":{"observed_at":"2026-05-15T02:39:33.662513Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.15234","last_updated":"2025-07-23T10:11:55Z","snapshot_observed_at":"2026-08-05T06:23:34.302204Z","submitted_at":"2023-12-23T11:57:53Z","title":"Towards Efficient Generative Large Language Model Serving: A Survey from Algorithms to Systems","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.15234","snapshot_observed_at":"2026-08-07T05:56:56.766216Z","title":"Towards efficient generative large language model serving: A survey from algorithms to systems,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.06579","last_updated":"2025-06-06T23:13:08Z","snapshot_observed_at":"2026-08-07T05:51:34.237166Z","submitted_at":"2025-06-06T23:13:08Z","title":"Towards Efficient Multi-LLM Inference: Characterization and Analysis of LLM Routing and Hierarchical Techniques","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-07T05:56:56.766216Z"},"links":{"cited_paper":"/paper/2312.15234","citing_paper":"/paper/2506.06579"},"observation_digest":"sha256:bf56eb565b045484000883db5b620d1a87073f976937af52f15ff5ef465afdc2","observation_id":"8a93df01-9df9-49f3-a6fd-8babc8be4a46","resolution":{"observed_at":"2026-08-07T05:56:56.766216Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.15234","last_updated":"2025-07-23T10:11:55Z","snapshot_observed_at":"2026-08-05T06:23:34.302204Z","submitted_at":"2023-12-23T11:57:53Z","title":"Towards Efficient Generative Large Language Model Serving: A Survey from Algorithms to Systems","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.15234","snapshot_observed_at":"2026-08-06T21:42:56.409360Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.23635","last_updated":"2025-06-30T09:04:25Z","snapshot_observed_at":"2026-08-06T21:33:17.286985Z","submitted_at":"2025-06-30T09:04:25Z","title":"Towards Building Private LLMs: Exploring Multi-Node Expert Parallelism on Apple Silicon for Mixture-of-Experts Large Language Model","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-06T21:42:56.409360Z"},"links":{"cited_paper":"/paper/2312.15234","citing_paper":"/paper/2506.23635"},"observation_digest":"sha256:891ceaa82c709a9af79e6b54d9c46273409e59e4b39500432ec5442639dc108d","observation_id":"65efe2eb-448e-4419-98a6-54677d2f9358","resolution":{"observed_at":"2026-08-06T21:42:56.409360Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.15234","last_updated":"2025-07-23T10:11:55Z","snapshot_observed_at":"2026-08-05T06:23:34.302204Z","submitted_at":"2023-12-23T11:57:53Z","title":"Towards Efficient Generative Large Language Model Serving: A Survey from Algorithms to Systems","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.15234","snapshot_observed_at":"2026-08-06T17:47:42.022798Z","title":"Towards efficient generative large language model serving: A survey from algorithms to systems,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.10026","last_updated":"2025-07-14T08:06:58Z","snapshot_observed_at":"2026-08-06T17:39:12.191621Z","submitted_at":"2025-07-14T08:06:58Z","title":"EAT: QoS-Aware Edge-Collaborative AIGC Task Scheduling via Attention-Guided Diffusion Reinforcement Learning","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-06T17:47:42.022798Z"},"links":{"cited_paper":"/paper/2312.15234","citing_paper":"/paper/2507.10026"},"observation_digest":"sha256:f5d4640601544887bf5f60f741e1fa0d58679c67bfd9ada14f6b9e6aabe22b01","observation_id":"cdb004f5-4ca8-47d6-bee4-81d164d3922f","resolution":{"observed_at":"2026-08-06T17:47:42.022798Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.15234","last_updated":"2025-07-23T10:11:55Z","snapshot_observed_at":"2026-08-05T06:23:34.302204Z","submitted_at":"2023-12-23T11:57:53Z","title":"Towards Efficient Generative Large Language Model Serving: A Survey from Algorithms to Systems","version":2},"cited_work":{"arxiv_id":"2312.15234","doi":null,"metadata_source":"pith","pith_arxiv_id":"2312.15234","snapshot_observed_at":"2026-07-10T00:56:40.976559Z","title":"Towards efficient generative large language model serving: A survey from algorithms to systems","venue":"cs.LG","work_id":"60d8ed33-eba3-401c-9124-609458817b15","year":2023},"citing_paper":{"arxiv_id":"2512.09427","last_updated":"2026-04-21T07:27:04Z","snapshot_observed_at":"2026-07-06T22:38:35.387503Z","submitted_at":"2025-12-10T08:52:20Z","title":"ODMA: On-Demand Memory Allocation Strategy for LLM Serving on LPDDR-Class Accelerators","version":5},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-05-16T23:54:08.052482Z"},"links":{"cited_paper":"/paper/2312.15234","citing_paper":"/paper/2512.09427"},"observation_digest":"sha256:7aa76254c2a72b25511162bbcb3315eba93423c14752ae19bed1ff3912f55a18","observation_id":"076456a4-3372-4a66-ba57-dd2621da37d8","resolution":{"observed_at":"2026-05-16T23:58:42.993560Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.15234","last_updated":"2025-07-23T10:11:55Z","snapshot_observed_at":"2026-08-05T06:23:34.302204Z","submitted_at":"2023-12-23T11:57:53Z","title":"Towards Efficient Generative Large Language Model Serving: A Survey from Algorithms to Systems","version":2},"cited_work":{"arxiv_id":"2312.15234","doi":null,"metadata_source":"pith","pith_arxiv_id":"2312.15234","snapshot_observed_at":"2026-07-10T00:56:40.976559Z","title":"Towards efficient generative large language model serving: A survey from algorithms to systems","venue":"cs.LG","work_id":"60d8ed33-eba3-401c-9124-609458817b15","year":2023},"citing_paper":{"arxiv_id":"2602.04163","last_updated":"2026-05-15T03:25:05Z","snapshot_observed_at":"2026-08-02T22:48:54.849826Z","submitted_at":"2026-02-04T02:54:37Z","title":"BPDQ: Bit-Plane Decomposition Quantization on a Variable Grid for Large Language Models","version":2},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-05-21T14:10:32.706531Z"},"links":{"cited_paper":"/paper/2312.15234","citing_paper":"/paper/2602.04163"},"observation_digest":"sha256:15daeac108a08a4df1956046ce25aeb46eb2569eff5a975757a6e044fc596285","observation_id":"2748bb5a-fc4d-4a3a-a71c-458bb389b015","resolution":{"observed_at":"2026-05-21T14:14:12.412025Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.15234","last_updated":"2025-07-23T10:11:55Z","snapshot_observed_at":"2026-08-05T06:23:34.302204Z","submitted_at":"2023-12-23T11:57:53Z","title":"Towards Efficient Generative Large Language Model Serving: A Survey from Algorithms to Systems","version":2},"cited_work":{"arxiv_id":"2312.15234","doi":null,"metadata_source":"pith","pith_arxiv_id":"2312.15234","snapshot_observed_at":"2026-07-10T00:56:40.976559Z","title":"Towards efficient generative large language model serving: A survey from algorithms to systems","venue":"cs.LG","work_id":"60d8ed33-eba3-401c-9124-609458817b15","year":2023},"citing_paper":{"arxiv_id":"2606.06453","last_updated":"2026-06-04T17:48:17Z","snapshot_observed_at":"2026-07-06T23:46:15.936608Z","submitted_at":"2026-06-04T17:48:17Z","title":"Vortex: Efficient and Programmable Sparse Attention Serving for AI Agents","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-06-28T01:07:14.691347Z"},"links":{"cited_paper":"/paper/2312.15234","citing_paper":"/paper/2606.06453"},"observation_digest":"sha256:f4ae8263a224f8eabf5889e4ccdd273cb2602e8aaef60670f4af01d70c4bb25d","observation_id":"513f6cc4-9748-489d-931b-1c2582440dbe","resolution":{"observed_at":"2026-07-02T13:36:59.470179Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.15234","last_updated":"2025-07-23T10:11:55Z","snapshot_observed_at":"2026-08-05T06:23:34.302204Z","submitted_at":"2023-12-23T11:57:53Z","title":"Towards Efficient Generative Large Language Model Serving: A Survey from Algorithms to Systems","version":2},"cited_work":{"arxiv_id":"2312.15234","doi":null,"metadata_source":"pith","pith_arxiv_id":"2312.15234","snapshot_observed_at":"2026-07-10T00:56:40.976559Z","title":"Towards efficient generative large language model serving: A survey from algorithms to systems","venue":"cs.LG","work_id":"60d8ed33-eba3-401c-9124-609458817b15","year":2023},"citing_paper":{"arxiv_id":"2606.21712","last_updated":"2026-06-19T19:56:21Z","snapshot_observed_at":"2026-08-06T10:39:09.239289Z","submitted_at":"2026-06-19T19:56:21Z","title":"BatchGen: An Architecture for Scalable and Efficient Batch Inference","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-06-26T12:58:25.319255Z"},"links":{"cited_paper":"/paper/2312.15234","citing_paper":"/paper/2606.21712"},"observation_digest":"sha256:e628886c3ac2393ff7c16139fed5863f03f23be8681dd698eac38123660e5bf5","observation_id":"ccb5aa73-3281-4d97-b438-63cad4299cc8","resolution":{"observed_at":"2026-07-04T07:39:39.560841Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.15234","last_updated":"2025-07-23T10:11:55Z","snapshot_observed_at":"2026-08-05T06:23:34.302204Z","submitted_at":"2023-12-23T11:57:53Z","title":"Towards Efficient Generative Large Language Model Serving: A Survey from Algorithms to Systems","version":2},"cited_work":{"arxiv_id":"2312.15234","doi":null,"metadata_source":"pith","pith_arxiv_id":"2312.15234","snapshot_observed_at":"2026-07-10T00:56:40.976559Z","title":"Towards efficient generative large language model serving: A survey from algorithms to systems","venue":"cs.LG","work_id":"60d8ed33-eba3-401c-9124-609458817b15","year":2023},"citing_paper":{"arxiv_id":"2607.08057","last_updated":"2026-07-09T02:11:18Z","snapshot_observed_at":"2026-08-04T11:12:32.355642Z","submitted_at":"2026-07-09T02:11:18Z","title":"Towards Efficient Large Language Model Serving: A Survey on System-Aware KV Cache Optimization","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-07-10T00:55:52.215700Z"},"links":{"cited_paper":"/paper/2312.15234","citing_paper":"/paper/2607.08057"},"observation_digest":"sha256:85398e74f245fa865e8d386b0fd41e0b423317152379d3e6841bc8c82bcbd87a","observation_id":"b1535c11-1425-4769-8eca-0e811aa67576","resolution":{"observed_at":"2026-07-10T00:56:40.977777Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2312.15234/citation-record","integrity":"/paper/2312.15234/integrity","json":"/paper/2312.15234/citation-record.json","paper":"/paper/2312.15234"},"outbound":[],"paper":{"arxiv_id":"2312.15234","last_updated":"2025-07-23T10:11:55Z","latest_version":2,"primary_category":"cs.LG","snapshot_observed_at":"2026-08-05T06:23:34.302204Z","submitted_at":"2023-12-23T11:57:53Z","title":"Towards Efficient Generative Large Language Model Serving: A Survey from Algorithms to Systems"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"thesis":"As of 7 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 10 inbound Pith citation observations for arXiv:2312.15234."}