{"as_of":"2026-08-09T02:07:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:4c12d28ee73743bb5e5492ab55eb774e30fc3976351ca048e16eb47d7d380d46","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":28,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":28,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-08T06:32:00.761636+00:00","state":"measured"},{"denominator":28,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":28,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-08T16:12:28.993285Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-04T18:50:04.384621Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2401.07851","last_updated":"2024-06-04T17:08:37Z","snapshot_observed_at":"2026-07-06T17:15:46.395218Z","submitted_at":"2024-01-15T17:26:50Z","title":"Unlocking Efficiency in Large Language Model Inference: A Comprehensive Survey of Speculative Decoding","version":3},"cited_work":{"arxiv_id":"2401.07851","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2401.07851","snapshot_observed_at":"2026-07-04T18:50:04.384621Z","title":"Unlocking efficiency in large language model in- ference: A comprehensive survey of speculative decoding","venue":null,"work_id":"1b0b6d28-5cea-4154-870b-534558ac4dcb","year":2024},"citing_paper":{"arxiv_id":"2402.03300","last_updated":"2024-04-27T15:25:53Z","snapshot_observed_at":"2026-08-06T14:58:42.911363Z","submitted_at":"2024-02-05T18:55:32Z","title":"DeepSeekMath: Pushing the Limits of Mathematical Reasoning in Open Language Models","version":3},"reference_index":54,"source":"arxiv_source","source_observed_at":"2026-05-24T03:23:18.827351Z"},"links":{"cited_paper":"/paper/2401.07851","citing_paper":"/paper/2402.03300"},"observation_digest":"sha256:b3792a3f310b222462f0a3c011d11a17cf86474b7007166289b269d0e4b5fa2d","observation_id":"e7c8b869-cec5-4ecb-ad09-032e8276c2cf","resolution":{"observed_at":"2026-05-24T03:23:49.645933Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.07851","last_updated":"2024-06-04T17:08:37Z","snapshot_observed_at":"2026-07-06T17:15:46.395218Z","submitted_at":"2024-01-15T17:26:50Z","title":"Unlocking Efficiency in Large Language Model Inference: A Comprehensive Survey of Speculative Decoding","version":3},"cited_work":{"arxiv_id":"2401.07851","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2401.07851","snapshot_observed_at":"2026-07-04T18:50:04.384621Z","title":"Unlocking efficiency in large language model in- ference: A comprehensive survey of speculative decoding","venue":null,"work_id":"1b0b6d28-5cea-4154-870b-534558ac4dcb","year":2024},"citing_paper":{"arxiv_id":"2404.14294","last_updated":"2024-07-19T04:47:36Z","snapshot_observed_at":"2026-07-06T18:03:47.096406Z","submitted_at":"2024-04-22T15:53:08Z","title":"A Survey on Efficient Inference for Large Language Models","version":3},"reference_index":269,"source":"pdf_text","source_observed_at":"2026-05-15T02:39:33.007894Z"},"links":{"cited_paper":"/paper/2401.07851","citing_paper":"/paper/2404.14294"},"observation_digest":"sha256:f1f67b3084b320594fd20e92e340345cda39aae74895bb86e2d274da17b86e4a","observation_id":"3d69fc85-6b0c-442c-b0d4-c8660aa56cb9","resolution":{"observed_at":"2026-05-15T02:39:33.456272Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.07851","last_updated":"2024-06-04T17:08:37Z","snapshot_observed_at":"2026-07-06T17:15:46.395218Z","submitted_at":"2024-01-15T17:26:50Z","title":"Unlocking Efficiency in Large Language Model Inference: A Comprehensive Survey of Speculative Decoding","version":3},"cited_work":{"arxiv_id":"2401.07851","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2401.07851","snapshot_observed_at":"2026-07-04T18:50:04.384621Z","title":"Unlocking efficiency in large language model in- ference: A comprehensive survey of speculative decoding","venue":null,"work_id":"1b0b6d28-5cea-4154-870b-534558ac4dcb","year":2024},"citing_paper":{"arxiv_id":"2501.05465","last_updated":"2026-05-14T16:52:31Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-01-03T19:53:57Z","title":"Small Language Models (SLMs) Can Still Pack a Punch: A survey (updated 2026)","version":2},"reference_index":137,"source":"pdf_text","source_observed_at":"2026-05-23T05:47:48.488826Z"},"links":{"cited_paper":"/paper/2401.07851","citing_paper":"/paper/2501.05465"},"observation_digest":"sha256:d30909727e42d25a478f18c7c33bf047b4378aaaf6804c39f7cbe1764211cecf","observation_id":"e84f72ca-173d-40ef-ae74-e116df59bb97","resolution":{"observed_at":"2026-05-23T05:52:37.555244Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.07851","last_updated":"2024-06-04T17:08:37Z","snapshot_observed_at":"2026-07-06T17:15:46.395218Z","submitted_at":"2024-01-15T17:26:50Z","title":"Unlocking Efficiency in Large Language Model Inference: A Comprehensive Survey of Speculative Decoding","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.07851","snapshot_observed_at":"2026-08-08T16:12:28.993285Z","title":"Unlocking efficiency in large language model inference: A comprehensive survey of speculative decoding","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2502.06282","last_updated":"2025-02-10T09:24:06Z","snapshot_observed_at":"2026-08-08T15:54:54.027148Z","submitted_at":"2025-02-10T09:24:06Z","title":"Jakiro: Boosting Speculative Decoding with Decoupled Multi-Head via MoE","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-08T16:12:28.993285Z"},"links":{"cited_paper":"/paper/2401.07851","citing_paper":"/paper/2502.06282"},"observation_digest":"sha256:3631d010bf87c8daf64df9a2af4a03606db0c1094843f51a5e14b5221a099901","observation_id":"8f738c5d-bf30-4c14-9d3d-1108f790f33a","resolution":{"observed_at":"2026-08-08T16:12:28.993285Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.07851","last_updated":"2024-06-04T17:08:37Z","snapshot_observed_at":"2026-07-06T17:15:46.395218Z","submitted_at":"2024-01-15T17:26:50Z","title":"Unlocking Efficiency in Large Language Model Inference: A Comprehensive Survey of Speculative Decoding","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.07851","snapshot_observed_at":"2026-08-08T11:11:44.653199Z","title":"Unlocking efficiency in large language model inference: A comprehensive survey of speculative decoding","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2502.08020","last_updated":"2025-03-19T16:26:10Z","snapshot_observed_at":"2026-08-08T11:04:51.000535Z","submitted_at":"2025-02-11T23:40:53Z","title":"Speculate, then Collaborate: Fusing Knowledge of Language Models during Decoding","version":2},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-08T11:11:44.653199Z"},"links":{"cited_paper":"/paper/2401.07851","citing_paper":"/paper/2502.08020"},"observation_digest":"sha256:32e3c3cbc66f5d1fa4cb5434d2e11b86ff7faefab292d6c2a87f686aa323fbf0","observation_id":"4b6a643e-a163-4897-818a-5763be994129","resolution":{"observed_at":"2026-08-08T11:11:44.653199Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.07851","last_updated":"2024-06-04T17:08:37Z","snapshot_observed_at":"2026-07-06T17:15:46.395218Z","submitted_at":"2024-01-15T17:26:50Z","title":"Unlocking Efficiency in Large Language Model Inference: A Comprehensive Survey of Speculative Decoding","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.07851","snapshot_observed_at":"2026-08-07T14:42:14.802495Z","title":"Unlocking efficiency in large language model inference: A comprehensive survey of speculative decoding","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.17941","last_updated":"2025-05-23T14:17:56Z","snapshot_observed_at":"2026-08-07T21:18:55.260268Z","submitted_at":"2025-05-23T14:17:56Z","title":"VeriThinker: Learning to Verify Makes Reasoning Model Efficient","version":1},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-08-07T14:42:14.802495Z"},"links":{"cited_paper":"/paper/2401.07851","citing_paper":"/paper/2505.17941"},"observation_digest":"sha256:56ac387da31aa2130a55ba11337bebb6c31f7917238c74d760ee93a6563e1b7e","observation_id":"351ee2cc-f9ba-452b-b620-bb0869482f2a","resolution":{"observed_at":"2026-08-07T14:42:14.802495Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.07851","last_updated":"2024-06-04T17:08:37Z","snapshot_observed_at":"2026-07-06T17:15:46.395218Z","submitted_at":"2024-01-15T17:26:50Z","title":"Unlocking Efficiency in Large Language Model Inference: A Comprehensive Survey of Speculative Decoding","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.07851","snapshot_observed_at":"2026-08-07T14:32:42.749845Z","title":"Unlocking efficiency in large language model inference: A comprehensive survey of speculative decoding","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.18629","last_updated":"2025-05-24T10:26:27Z","snapshot_observed_at":"2026-08-08T01:06:18.417714Z","submitted_at":"2025-05-24T10:26:27Z","title":"Think Before You Accept: Semantic Reflective Verification for Faster Speculative Decoding","version":1},"reference_index":28,"source":"arxiv_source","source_observed_at":"2026-08-07T14:32:42.749845Z"},"links":{"cited_paper":"/paper/2401.07851","citing_paper":"/paper/2505.18629"},"observation_digest":"sha256:89007eda71c1cfab94f8ca79e40d21dc8a420e3b7de781a66d68cc25532fca54","observation_id":"3aea8502-c3da-49e5-bd95-5429f5c2cbae","resolution":{"observed_at":"2026-08-07T14:32:42.749845Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.07851","last_updated":"2024-06-04T17:08:37Z","snapshot_observed_at":"2026-07-06T17:15:46.395218Z","submitted_at":"2024-01-15T17:26:50Z","title":"Unlocking Efficiency in Large Language Model Inference: A Comprehensive Survey of Speculative Decoding","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.07851","snapshot_observed_at":"2026-08-07T11:30:57.001513Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.02391","last_updated":"2025-06-03T03:13:27Z","snapshot_observed_at":"2026-08-08T01:16:04.126035Z","submitted_at":"2025-06-03T03:13:27Z","title":"Consultant Decoding: Yet Another Synergistic Mechanism","version":1},"reference_index":31,"source":"arxiv_source","source_observed_at":"2026-08-07T11:30:57.001513Z"},"links":{"cited_paper":"/paper/2401.07851","citing_paper":"/paper/2506.02391"},"observation_digest":"sha256:d99bb0d0bee55157c955110e39d38aa6d9cb524db3d235a501597490c6b75286","observation_id":"f9be1957-6551-4f7e-b129-df13e605b179","resolution":{"observed_at":"2026-08-07T11:30:57.001513Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.07851","last_updated":"2024-06-04T17:08:37Z","snapshot_observed_at":"2026-07-06T17:15:46.395218Z","submitted_at":"2024-01-15T17:26:50Z","title":"Unlocking Efficiency in Large Language Model Inference: A Comprehensive Survey of Speculative Decoding","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.07851","snapshot_observed_at":"2026-08-07T11:02:39.384667Z","title":"Unlocking efficiency in large language model inference: A comprehensive survey of speculative decoding","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.03700","last_updated":"2025-06-04T08:32:30Z","snapshot_observed_at":"2026-08-08T01:16:24.422333Z","submitted_at":"2025-06-04T08:32:30Z","title":"AdaDecode: Accelerating LLM Decoding with Adaptive Layer Parallelism","version":1},"reference_index":63,"source":"arxiv_source","source_observed_at":"2026-08-07T11:02:39.384667Z"},"links":{"cited_paper":"/paper/2401.07851","citing_paper":"/paper/2506.03700"},"observation_digest":"sha256:1b2da0d983384704651370fa56abf9a8e8c9e15569d0d4d14a6d6b92fb4b7c74","observation_id":"bc2d3b77-700d-4849-9bce-2a0b5b5c1298","resolution":{"observed_at":"2026-08-07T11:02:39.384667Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.07851","last_updated":"2024-06-04T17:08:37Z","snapshot_observed_at":"2026-07-06T17:15:46.395218Z","submitted_at":"2024-01-15T17:26:50Z","title":"Unlocking Efficiency in Large Language Model Inference: A Comprehensive Survey of Speculative Decoding","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.07851","snapshot_observed_at":"2026-08-07T00:22:46.418338Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.14158","last_updated":"2025-06-17T03:38:19Z","snapshot_observed_at":"2026-08-09T00:36:23.093929Z","submitted_at":"2025-06-17T03:38:19Z","title":"S$^4$C: Speculative Sampling with Syntactic and Semantic Coherence for Efficient Inference of Large Language Models","version":1},"reference_index":27,"source":"arxiv_source","source_observed_at":"2026-08-07T00:22:46.418338Z"},"links":{"cited_paper":"/paper/2401.07851","citing_paper":"/paper/2506.14158"},"observation_digest":"sha256:71aa0428ae214f422d1e759a570c7434d4b66ed684bb48b631cc5c7c1b7a8678","observation_id":"03a5e258-919c-4c63-9356-739963217701","resolution":{"observed_at":"2026-08-07T00:22:46.418338Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.07851","last_updated":"2024-06-04T17:08:37Z","snapshot_observed_at":"2026-07-06T17:15:46.395218Z","submitted_at":"2024-01-15T17:26:50Z","title":"Unlocking Efficiency in Large Language Model Inference: A Comprehensive Survey of Speculative Decoding","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.07851","snapshot_observed_at":"2026-08-06T14:17:44.088718Z","title":"Unlocking efficiency in large language model infer- ence: A comprehensive survey of speculative decoding,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.19608","last_updated":"2025-07-25T18:23:18Z","snapshot_observed_at":"2026-08-06T14:17:43.357204Z","submitted_at":"2025-07-25T18:23:18Z","title":"DeltaLLM: A Training-Free Framework Exploiting Temporal Sparsity for Efficient Edge LLM Inference","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-06T14:17:44.088718Z"},"links":{"cited_paper":"/paper/2401.07851","citing_paper":"/paper/2507.19608"},"observation_digest":"sha256:f2899a02275643bcc89080c44387cf1dc35a4b820ae9767917ff3b8965854291","observation_id":"98267ba7-8eeb-4aeb-9e78-24ce97bd8e69","resolution":{"observed_at":"2026-08-06T14:17:44.088718Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.07851","last_updated":"2024-06-04T17:08:37Z","snapshot_observed_at":"2026-07-06T17:15:46.395218Z","submitted_at":"2024-01-15T17:26:50Z","title":"Unlocking Efficiency in Large Language Model Inference: A Comprehensive Survey of Speculative Decoding","version":3},"cited_work":{"arxiv_id":"2401.07851","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2401.07851","snapshot_observed_at":"2026-07-04T18:50:04.384621Z","title":"Unlocking efficiency in large language model in- ference: A comprehensive survey of speculative decoding","venue":null,"work_id":"1b0b6d28-5cea-4154-870b-534558ac4dcb","year":2024},"citing_paper":{"arxiv_id":"2602.06932","last_updated":"2026-05-02T06:02:57Z","snapshot_observed_at":"2026-08-06T14:10:22.450479Z","submitted_at":"2026-02-06T18:28:54Z","title":"When RL Meets Adaptive Speculative Training: A Unified Training-Serving System","version":3},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-05-16T06:33:41.860803Z"},"links":{"cited_paper":"/paper/2401.07851","citing_paper":"/paper/2602.06932"},"observation_digest":"sha256:72076b3a5f375121f35b3ee40b5dc7db73f2c29b2667317e57cb6168b1b6708b","observation_id":"4e49a438-df9f-4712-870a-40c447917794","resolution":{"observed_at":"2026-05-16T06:37:28.884383Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.07851","last_updated":"2024-06-04T17:08:37Z","snapshot_observed_at":"2026-07-06T17:15:46.395218Z","submitted_at":"2024-01-15T17:26:50Z","title":"Unlocking Efficiency in Large Language Model Inference: A Comprehensive Survey of Speculative Decoding","version":3},"cited_work":{"arxiv_id":"2401.07851","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2401.07851","snapshot_observed_at":"2026-07-04T18:50:04.384621Z","title":"Unlocking efficiency in large language model in- ference: A comprehensive survey of speculative decoding","venue":null,"work_id":"1b0b6d28-5cea-4154-870b-534558ac4dcb","year":2024},"citing_paper":{"arxiv_id":"2604.09603","last_updated":"2026-05-14T06:18:54Z","snapshot_observed_at":"2026-07-31T19:11:06.067090Z","submitted_at":"2026-03-10T03:51:24Z","title":"ECHO: Elastic Speculative Decoding with Sparse Gating for High-Concurrency Scenarios","version":2},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-05-15T14:09:59.938676Z"},"links":{"cited_paper":"/paper/2401.07851","citing_paper":"/paper/2604.09603"},"observation_digest":"sha256:7542904128fb62c79d7ecf5db8951eb2953056cbec0c04b0b741da45f8a6c112","observation_id":"6291821a-e0de-4494-8be3-44ba6674a1fc","resolution":{"observed_at":"2026-05-15T14:10:03.112922Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.07851","last_updated":"2024-06-04T17:08:37Z","snapshot_observed_at":"2026-07-06T17:15:46.395218Z","submitted_at":"2024-01-15T17:26:50Z","title":"Unlocking Efficiency in Large Language Model Inference: A Comprehensive Survey of Speculative Decoding","version":3},"cited_work":{"arxiv_id":"2401.07851","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2401.07851","snapshot_observed_at":"2026-07-04T18:50:04.384621Z","title":"Unlocking efficiency in large language model in- ference: A comprehensive survey of speculative decoding","venue":null,"work_id":"1b0b6d28-5cea-4154-870b-534558ac4dcb","year":2024},"citing_paper":{"arxiv_id":"2604.20503","last_updated":"2026-04-22T12:44:39Z","snapshot_observed_at":"2026-07-06T23:06:55.880484Z","submitted_at":"2026-04-22T12:44:39Z","title":"FASER: Fine-Grained Phase Management for Speculative Decoding in Dynamic LLM Serving","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-05-09T22:56:19.734262Z"},"links":{"cited_paper":"/paper/2401.07851","citing_paper":"/paper/2604.20503"},"observation_digest":"sha256:41aa5cce4c00ab8ba2b36084ba879b534503d870a61a2597c583141788304080","observation_id":"bbc09dcd-f2d6-4dd4-8436-93db38143891","resolution":{"observed_at":"2026-05-09T22:59:17.685404Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.07851","last_updated":"2024-06-04T17:08:37Z","snapshot_observed_at":"2026-07-06T17:15:46.395218Z","submitted_at":"2024-01-15T17:26:50Z","title":"Unlocking Efficiency in Large Language Model Inference: A Comprehensive Survey of Speculative Decoding","version":3},"cited_work":{"arxiv_id":"2401.07851","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2401.07851","snapshot_observed_at":"2026-07-04T18:50:04.384621Z","title":"Unlocking efficiency in large language model in- ference: A comprehensive survey of speculative decoding","venue":null,"work_id":"1b0b6d28-5cea-4154-870b-534558ac4dcb","year":2024},"citing_paper":{"arxiv_id":"2604.22906","last_updated":"2026-04-24T16:56:53Z","snapshot_observed_at":"2026-08-07T20:25:38.013655Z","submitted_at":"2026-04-24T16:56:53Z","title":"Network Edge Inference for Large Language Models: Principles, Techniques, and Opportunities","version":1},"reference_index":169,"source":"pdf_text","source_observed_at":"2026-05-08T09:45:57.201837Z"},"links":{"cited_paper":"/paper/2401.07851","citing_paper":"/paper/2604.22906"},"observation_digest":"sha256:13d1218f7d6c1eabad9abd2ef3c107c1f3ce1e8f48abc758caebed601c95ddad","observation_id":"e609ecc1-d828-49f8-8e06-529c24ee3a5a","resolution":{"observed_at":"2026-05-11T20:16:10.385717Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.07851","last_updated":"2024-06-04T17:08:37Z","snapshot_observed_at":"2026-07-06T17:15:46.395218Z","submitted_at":"2024-01-15T17:26:50Z","title":"Unlocking Efficiency in Large Language Model Inference: A Comprehensive Survey of Speculative Decoding","version":3},"cited_work":{"arxiv_id":"2401.07851","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2401.07851","snapshot_observed_at":"2026-07-04T18:50:04.384621Z","title":"Unlocking efficiency in large language model in- ference: A comprehensive survey of speculative decoding","venue":null,"work_id":"1b0b6d28-5cea-4154-870b-534558ac4dcb","year":2024},"citing_paper":{"arxiv_id":"2604.26412","last_updated":"2026-05-09T06:52:02Z","snapshot_observed_at":"2026-07-06T23:12:06.349884Z","submitted_at":"2026-04-29T08:25:01Z","title":"When Hidden States Drift: Can KV Caches Rescue Long-Range Speculative Decoding?","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-05-07T11:08:25.132013Z"},"links":{"cited_paper":"/paper/2401.07851","citing_paper":"/paper/2604.26412"},"observation_digest":"sha256:48ee2587622ad1faab2550a1ce8fe3b39cfb729ec76eb66ed2ab28885daba802","observation_id":"80a3d0c8-c0ae-4104-a2ef-b64929f2a1a5","resolution":{"observed_at":"2026-05-12T09:21:26.630154Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.07851","last_updated":"2024-06-04T17:08:37Z","snapshot_observed_at":"2026-07-06T17:15:46.395218Z","submitted_at":"2024-01-15T17:26:50Z","title":"Unlocking Efficiency in Large Language Model Inference: A Comprehensive Survey of Speculative Decoding","version":3},"cited_work":{"arxiv_id":"2401.07851","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2401.07851","snapshot_observed_at":"2026-07-04T18:50:04.384621Z","title":"Unlocking efficiency in large language model in- ference: A comprehensive survey of speculative decoding","venue":null,"work_id":"1b0b6d28-5cea-4154-870b-534558ac4dcb","year":2024},"citing_paper":{"arxiv_id":"2604.26412","last_updated":"2026-05-09T06:52:02Z","snapshot_observed_at":"2026-07-06T23:12:06.349884Z","submitted_at":"2026-04-29T08:25:01Z","title":"When Hidden States Drift: Can KV Caches Rescue Long-Range Speculative Decoding?","version":2},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-05-12T00:59:11.690911Z"},"links":{"cited_paper":"/paper/2401.07851","citing_paper":"/paper/2604.26412"},"observation_digest":"sha256:0cb2664a1cf17b48bf7eb580d06539f1315d2e63dc7be1f75b902d67e2065e90","observation_id":"45b71c1b-b261-4f19-8aba-346a2c817270","resolution":{"observed_at":"2026-05-12T08:36:23.953655Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.07851","last_updated":"2024-06-04T17:08:37Z","snapshot_observed_at":"2026-07-06T17:15:46.395218Z","submitted_at":"2024-01-15T17:26:50Z","title":"Unlocking Efficiency in Large Language Model Inference: A Comprehensive Survey of Speculative Decoding","version":3},"cited_work":{"arxiv_id":"2401.07851","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2401.07851","snapshot_observed_at":"2026-07-04T18:50:04.384621Z","title":"Unlocking efficiency in large language model in- ference: A comprehensive survey of speculative decoding","venue":null,"work_id":"1b0b6d28-5cea-4154-870b-534558ac4dcb","year":2024},"citing_paper":{"arxiv_id":"2605.10195","last_updated":"2026-05-14T07:42:56Z","snapshot_observed_at":"2026-08-03T09:42:12.047946Z","submitted_at":"2026-05-11T08:45:17Z","title":"Breaking the Reward Barrier: Accelerating Tree-of-Thought Reasoning via Speculative Exploration","version":1},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-05-12T03:51:52.375703Z"},"links":{"cited_paper":"/paper/2401.07851","citing_paper":"/paper/2605.10195"},"observation_digest":"sha256:d6470a798aaebd743294e4675d8a04e649a18f825bba20853b60154df51dfc1d","observation_id":"c3f4aeeb-d47d-416f-8446-7e6dc91f13e3","resolution":{"observed_at":"2026-05-12T06:51:30.281087Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.07851","last_updated":"2024-06-04T17:08:37Z","snapshot_observed_at":"2026-07-06T17:15:46.395218Z","submitted_at":"2024-01-15T17:26:50Z","title":"Unlocking Efficiency in Large Language Model Inference: A Comprehensive Survey of Speculative Decoding","version":3},"cited_work":{"arxiv_id":"2401.07851","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2401.07851","snapshot_observed_at":"2026-07-04T18:50:04.384621Z","title":"Unlocking efficiency in large language model in- ference: A comprehensive survey of speculative decoding","venue":null,"work_id":"1b0b6d28-5cea-4154-870b-534558ac4dcb","year":2024},"citing_paper":{"arxiv_id":"2605.10195","last_updated":"2026-05-14T07:42:56Z","snapshot_observed_at":"2026-08-03T09:42:12.047946Z","submitted_at":"2026-05-11T08:45:17Z","title":"Breaking the Reward Barrier: Accelerating Tree-of-Thought Reasoning via Speculative Exploration","version":2},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-05-15T05:11:32.053440Z"},"links":{"cited_paper":"/paper/2401.07851","citing_paper":"/paper/2605.10195"},"observation_digest":"sha256:fa3f4331f474907c46829411802cc09fc9547de218faac3a8bac4ac1f2bbb9b9","observation_id":"8313ebec-ae47-469f-8d9b-6f9299776a4f","resolution":{"observed_at":"2026-05-15T05:15:03.451418Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.07851","last_updated":"2024-06-04T17:08:37Z","snapshot_observed_at":"2026-07-06T17:15:46.395218Z","submitted_at":"2024-01-15T17:26:50Z","title":"Unlocking Efficiency in Large Language Model Inference: A Comprehensive Survey of Speculative Decoding","version":3},"cited_work":{"arxiv_id":"2401.07851","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2401.07851","snapshot_observed_at":"2026-07-04T18:50:04.384621Z","title":"Unlocking efficiency in large language model in- ference: A comprehensive survey of speculative decoding","venue":null,"work_id":"1b0b6d28-5cea-4154-870b-534558ac4dcb","year":2024},"citing_paper":{"arxiv_id":"2605.18810","last_updated":"2026-05-12T06:27:57Z","snapshot_observed_at":"2026-08-03T22:19:38.338198Z","submitted_at":"2026-05-12T06:27:57Z","title":"D-PACE: Dynamic Position-Aware Cross-Entropy for Parallel Speculative Drafting","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-05-20T21:56:42.264380Z"},"links":{"cited_paper":"/paper/2401.07851","citing_paper":"/paper/2605.18810"},"observation_digest":"sha256:03f6b98f29a55fa549fae1441e32a801c38c0122dff437a18ba89a82b2ea30d9","observation_id":"86141a99-b4a2-44e1-9db8-182f15452ebe","resolution":{"observed_at":"2026-05-20T21:59:06.117035Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.07851","last_updated":"2024-06-04T17:08:37Z","snapshot_observed_at":"2026-07-06T17:15:46.395218Z","submitted_at":"2024-01-15T17:26:50Z","title":"Unlocking Efficiency in Large Language Model Inference: A Comprehensive Survey of Speculative Decoding","version":3},"cited_work":{"arxiv_id":"2401.07851","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2401.07851","snapshot_observed_at":"2026-07-04T18:50:04.384621Z","title":"Unlocking efficiency in large language model in- ference: A comprehensive survey of speculative decoding","venue":null,"work_id":"1b0b6d28-5cea-4154-870b-534558ac4dcb","year":2024},"citing_paper":{"arxiv_id":"2605.20104","last_updated":"2026-05-19T16:55:48Z","snapshot_observed_at":"2026-08-04T03:30:21.702008Z","submitted_at":"2026-05-19T16:55:48Z","title":"Draft Less, Retrieve More: Hybrid Tree Construction for Speculative Decoding","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-05-20T06:58:07.996335Z"},"links":{"cited_paper":"/paper/2401.07851","citing_paper":"/paper/2605.20104"},"observation_digest":"sha256:4f86f0bfba114cb2039cbb70d3d13facacf6266bb600d65de5280902a48b8e09","observation_id":"8b7018ec-e240-42a5-84ee-d0a9083ec63c","resolution":{"observed_at":"2026-05-20T07:03:06.426899Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.07851","last_updated":"2024-06-04T17:08:37Z","snapshot_observed_at":"2026-07-06T17:15:46.395218Z","submitted_at":"2024-01-15T17:26:50Z","title":"Unlocking Efficiency in Large Language Model Inference: A Comprehensive Survey of Speculative Decoding","version":3},"cited_work":{"arxiv_id":"2401.07851","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2401.07851","snapshot_observed_at":"2026-07-04T18:50:04.384621Z","title":"Unlocking efficiency in large language model in- ference: A comprehensive survey of speculative decoding","venue":null,"work_id":"1b0b6d28-5cea-4154-870b-534558ac4dcb","year":2024},"citing_paper":{"arxiv_id":"2605.29639","last_updated":"2026-05-28T09:07:06Z","snapshot_observed_at":"2026-08-03T07:14:27.185639Z","submitted_at":"2026-05-28T09:07:06Z","title":"RTP-LLM: High-Performance Alibaba LLM Inference Engine","version":1},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-06-28T23:52:40.763228Z"},"links":{"cited_paper":"/paper/2401.07851","citing_paper":"/paper/2605.29639"},"observation_digest":"sha256:f349bdbd3b767690c7664a5ee32f11bec434de2fef7eec50f9bdeb92ed00b40c","observation_id":"def9a5c9-d4c8-4a89-af31-72992f9fc9d2","resolution":{"observed_at":"2026-06-28T23:52:49.154232Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.07851","last_updated":"2024-06-04T17:08:37Z","snapshot_observed_at":"2026-07-06T17:15:46.395218Z","submitted_at":"2024-01-15T17:26:50Z","title":"Unlocking Efficiency in Large Language Model Inference: A Comprehensive Survey of Speculative Decoding","version":3},"cited_work":{"arxiv_id":"2401.07851","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2401.07851","snapshot_observed_at":"2026-07-04T18:50:04.384621Z","title":"Unlocking efficiency in large language model in- ference: A comprehensive survey of speculative decoding","venue":null,"work_id":"1b0b6d28-5cea-4154-870b-534558ac4dcb","year":2024},"citing_paper":{"arxiv_id":"2606.04921","last_updated":"2026-06-03T14:17:12Z","snapshot_observed_at":"2026-07-06T23:44:57.208389Z","submitted_at":"2026-06-03T14:17:12Z","title":"SURF: Separation via Unsupervised Remixing Flow","version":1},"reference_index":185,"source":"arxiv_source","source_observed_at":"2026-06-28T05:07:10.235599Z"},"links":{"cited_paper":"/paper/2401.07851","citing_paper":"/paper/2606.04921"},"observation_digest":"sha256:92908bd74cff3349883b63971372e66e9dfee28fa5af31ddfb20d25df0ac09b0","observation_id":"eddf3414-193d-46a8-ac71-c15c7f02f5c3","resolution":{"observed_at":"2026-07-02T10:16:52.609778Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.07851","last_updated":"2024-06-04T17:08:37Z","snapshot_observed_at":"2026-07-06T17:15:46.395218Z","submitted_at":"2024-01-15T17:26:50Z","title":"Unlocking Efficiency in Large Language Model Inference: A Comprehensive Survey of Speculative Decoding","version":3},"cited_work":{"arxiv_id":"2401.07851","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2401.07851","snapshot_observed_at":"2026-07-04T18:50:04.384621Z","title":"Unlocking efficiency in large language model in- ference: A comprehensive survey of speculative decoding","venue":null,"work_id":"1b0b6d28-5cea-4154-870b-534558ac4dcb","year":2024},"citing_paper":{"arxiv_id":"2606.25091","last_updated":"2026-06-23T18:55:18Z","snapshot_observed_at":"2026-08-06T04:30:21.060187Z","submitted_at":"2026-06-23T18:55:18Z","title":"Speculation at a Distance: Where Edge-Cloud Speculative Decoding Actually Pays Off","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-06-25T22:31:27.598322Z"},"links":{"cited_paper":"/paper/2401.07851","citing_paper":"/paper/2606.25091"},"observation_digest":"sha256:b3c49b15758dd977bb1ce6acdaeefb7bfa1b08be3dee981ce7c3e9520e1bd78c","observation_id":"03d59875-df8e-4692-a416-a10eb5d24238","resolution":{"observed_at":"2026-07-04T18:50:04.386306Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-08T06:32:00.761636+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.07851","last_updated":"2024-06-04T17:08:37Z","snapshot_observed_at":"2026-07-06T17:15:46.395218Z","submitted_at":"2024-01-15T17:26:50Z","title":"Unlocking Efficiency in Large Language Model Inference: A Comprehensive Survey of Speculative Decoding","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.07851","snapshot_observed_at":"2026-08-02T01:34:46.570419Z","title":"arXiv preprint arXiv:2401.07851 , year =","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.14647","last_updated":"2026-07-16T07:18:05Z","snapshot_observed_at":"2026-08-09T00:39:37.122357Z","submitted_at":"2026-07-16T07:18:05Z","title":"D-cut: Adaptive Verification Depth Pruning for Batched Speculative Decoding","version":1},"reference_index":60,"source":"arxiv_source","source_observed_at":"2026-08-02T01:34:46.570419Z"},"links":{"cited_paper":"/paper/2401.07851","citing_paper":"/paper/2607.14647"},"observation_digest":"sha256:9fea2cf8e554a77ca2fcf220f02d14f0fca92b61275bfabafb5d2c0f41a85be6","observation_id":"9a768270-878e-4961-9570-8d80e02cc279","resolution":{"observed_at":"2026-08-02T01:34:46.570419Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.07851","last_updated":"2024-06-04T17:08:37Z","snapshot_observed_at":"2026-07-06T17:15:46.395218Z","submitted_at":"2024-01-15T17:26:50Z","title":"Unlocking Efficiency in Large Language Model Inference: A Comprehensive Survey of Speculative Decoding","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.07851","snapshot_observed_at":"2026-08-01T10:14:14.991489Z","title":"arXiv:2401.07851","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.20327","last_updated":"2026-07-22T16:14:26Z","snapshot_observed_at":"2026-08-07T18:49:46.472241Z","submitted_at":"2026-07-22T16:14:26Z","title":"PyroDash: Cost-Efficient Token-Level Small-Large Language Model Collaborative Inference","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-01T10:14:14.991489Z"},"links":{"cited_paper":"/paper/2401.07851","citing_paper":"/paper/2607.20327"},"observation_digest":"sha256:2dea8d93863527ea7219fcc5c0d2f4ec9ab4aa72bea4c5a3a6e50d7324201d48","observation_id":"3d44352d-c718-4c95-81c8-019008783cb1","resolution":{"observed_at":"2026-08-01T10:14:14.991489Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.07851","last_updated":"2024-06-04T17:08:37Z","snapshot_observed_at":"2026-07-06T17:15:46.395218Z","submitted_at":"2024-01-15T17:26:50Z","title":"Unlocking Efficiency in Large Language Model Inference: A Comprehensive Survey of Speculative Decoding","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.07851","snapshot_observed_at":"2026-08-01T01:22:13.844047Z","title":"Yunfan Xiong, Ruoyu Zhang, Yanzeng Li, Tianhao Wu, and Lei Zou","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.25852","last_updated":"2026-07-29T07:31:33Z","snapshot_observed_at":"2026-08-07T17:12:03.331999Z","submitted_at":"2026-07-28T15:25:05Z","title":"AngelSpec: Towards Real-World High Performance Inference with Speculative Decoding","version":2},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-01T01:22:13.844047Z"},"links":{"cited_paper":"/paper/2401.07851","citing_paper":"/paper/2607.25852"},"observation_digest":"sha256:a4b685ee01080a87e7ffc6fb40272b97d9493039ccae7ce1c15b9ee6baf90c2d","observation_id":"9b1e2b73-e24a-4ae4-9db5-7ddffd24fded","resolution":{"observed_at":"2026-08-01T01:22:13.844047Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.07851","last_updated":"2024-06-04T17:08:37Z","snapshot_observed_at":"2026-07-06T17:15:46.395218Z","submitted_at":"2024-01-15T17:26:50Z","title":"Unlocking Efficiency in Large Language Model Inference: A Comprehensive Survey of Speculative Decoding","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.07851","snapshot_observed_at":"2026-08-01T12:06:58.268790Z","title":"arXiv preprint arXiv:2401.07851 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.26627","last_updated":"2026-07-29T08:54:27Z","snapshot_observed_at":"2026-08-06T18:27:46.666122Z","submitted_at":"2026-07-29T08:54:27Z","title":"Revisiting Lossy Verification in Speculative Decoding: Mechanisms, Trade-offs, and Failure Modes","version":1},"reference_index":16,"source":"arxiv_source","source_observed_at":"2026-08-01T12:06:58.268790Z"},"links":{"cited_paper":"/paper/2401.07851","citing_paper":"/paper/2607.26627"},"observation_digest":"sha256:4fc70dc11b799ab9f29ef09e58b3b950011011c5adb8cc1ad6686e622f598181","observation_id":"5389dc48-786e-454d-9406-0d90fd08b364","resolution":{"observed_at":"2026-08-01T12:06:58.268790Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2401.07851/citation-record","integrity":"/paper/2401.07851/integrity","json":"/paper/2401.07851/citation-record.json","paper":"/paper/2401.07851"},"outbound":[],"paper":{"arxiv_id":"2401.07851","last_updated":"2024-06-04T17:08:37Z","latest_version":3,"primary_category":"cs.CL","snapshot_observed_at":"2026-07-06T17:15:46.395218Z","submitted_at":"2024-01-15T17:26:50Z","title":"Unlocking Efficiency in Large Language Model Inference: A Comprehensive Survey of Speculative Decoding"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-08T06:32:00.761636+00:00","source":"crossref"},{"observed_at":"2026-08-08T06:31:55.24221+00:00","source":"retraction_watch"}],"thesis":"As of 9 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 28 inbound Pith citation observations for arXiv:2401.07851."}