{"as_of":"2026-08-17T22:41:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:5b6f265cef357d64447f6028d19bc958297e1ee486d4efd588dc108b002cebd7","coverage":[{"denominator":22,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":22,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-05T13:48:23.144764Z","state":"measured"},{"denominator":22,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":22,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-17T06:30:58.91139+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2509.04474/citation-record","integrity":"/paper/2509.04474/integrity","json":"/paper/2509.04474/citation-record.json","paper":"/paper/2509.04474"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2302.01318","last_updated":"2023-02-02T18:44:11Z","snapshot_observed_at":"2026-08-13T23:55:57.074762Z","submitted_at":"2023-02-02T18:44:11Z","title":"Accelerating Large Language Model Decoding with Speculative Sampling","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2302.01318","snapshot_observed_at":"2026-08-05T13:48:21.159265Z","title":"Accelerating large language model decoding with speculative sampling","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2509.04474","last_updated":"2025-08-30T01:54:55Z","snapshot_observed_at":"2026-08-17T14:22:16.859857Z","submitted_at":"2025-08-30T01:54:55Z","title":"Scaling Up, Speeding Up: A Benchmark of Speculative Decoding for Efficient LLM Test-Time Scaling","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-05T13:48:21.159265Z"},"links":{"cited_paper":"/paper/2302.01318","citing_paper":"/paper/2509.04474"},"observation_digest":"sha256:10603f7727b5dc96a8298b8faa9d3e7a6fd0e69221bf1fe241d61bad1f45c7cc","observation_id":"c0f3bf10-0159-4c65-a2a0-109be6e5d20d","resolution":{"observed_at":"2026-08-05T13:48:21.159265Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.06769","last_updated":"2025-11-03T00:53:34Z","snapshot_observed_at":"2026-08-17T06:12:37.525438Z","submitted_at":"2024-12-09T18:55:56Z","title":"Training Large Language Models to Reason in a Continuous Latent Space","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.06769","snapshot_observed_at":"2026-08-05T13:48:21.516547Z","title":"Training large language models to reason in a continuous latent space","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2509.04474","last_updated":"2025-08-30T01:54:55Z","snapshot_observed_at":"2026-08-17T14:22:16.859857Z","submitted_at":"2025-08-30T01:54:55Z","title":"Scaling Up, Speeding Up: A Benchmark of Speculative Decoding for Efficient LLM Test-Time Scaling","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-05T13:48:21.516547Z"},"links":{"cited_paper":"/paper/2412.06769","citing_paper":"/paper/2509.04474"},"observation_digest":"sha256:4e22754dc424d3c934b3ae6875bc8ddb0b38c3f4fbe07f3abba4a872c7d22330","observation_id":"fda6280c-cd33-442d-8336-8d4c82be5159","resolution":{"observed_at":"2026-08-05T13:48:21.516547Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.01840","last_updated":"2025-04-23T07:08:17Z","snapshot_observed_at":"2026-08-13T11:30:48.832931Z","submitted_at":"2025-03-03T18:59:04Z","title":"EAGLE-3: Scaling up Inference Acceleration of Large Language Models via Training-Time Test","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.01840","snapshot_observed_at":"2026-08-05T13:48:21.574388Z","title":"Eagle-2: Faster inference of language models with dynamic draft trees","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2509.04474","last_updated":"2025-08-30T01:54:55Z","snapshot_observed_at":"2026-08-17T14:22:16.859857Z","submitted_at":"2025-08-30T01:54:55Z","title":"Scaling Up, Speeding Up: A Benchmark of Speculative Decoding for Efficient LLM Test-Time Scaling","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-05T13:48:21.574388Z"},"links":{"cited_paper":"/paper/2503.01840","citing_paper":"/paper/2509.04474"},"observation_digest":"sha256:735733e7eeda186b2d5de9a087d2d8c5550379dea6e43cbd3a0af0b16124e789","observation_id":"97d15c73-75eb-468f-b5a4-b949fa073faf","resolution":{"observed_at":"2026-08-05T13:48:21.574388Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2203.07814","last_updated":"2022-02-08T23:16:31Z","snapshot_observed_at":"2026-08-15T05:41:03.327624Z","submitted_at":"2022-02-08T23:16:31Z","title":"Competition-Level Code Generation with AlphaCode","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2203.07814","snapshot_observed_at":"2026-08-05T13:48:21.656281Z","title":"Competition-level code generation with alphacode","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2509.04474","last_updated":"2025-08-30T01:54:55Z","snapshot_observed_at":"2026-08-17T14:22:16.859857Z","submitted_at":"2025-08-30T01:54:55Z","title":"Scaling Up, Speeding Up: A Benchmark of Speculative Decoding for Efficient LLM Test-Time Scaling","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-05T13:48:21.656281Z"},"links":{"cited_paper":"/paper/2203.07814","citing_paper":"/paper/2509.04474"},"observation_digest":"sha256:df80c89dc35372263c2557e52fe7e86e0e40857cb3ca60f78ea0b04e746235bf","observation_id":"e8487933-93ff-4006-9f1e-1d273b3ac010","resolution":{"observed_at":"2026-08-05T13:48:21.656281Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.20050","last_updated":"2023-05-31T17:24:00Z","snapshot_observed_at":"2026-08-17T09:42:34.746112Z","submitted_at":"2023-05-31T17:24:00Z","title":"Let's Verify Step by Step","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.20050","snapshot_observed_at":"2026-08-05T13:48:21.885995Z","title":"Let’s verify step by step","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2509.04474","last_updated":"2025-08-30T01:54:55Z","snapshot_observed_at":"2026-08-17T14:22:16.859857Z","submitted_at":"2025-08-30T01:54:55Z","title":"Scaling Up, Speeding Up: A Benchmark of Speculative Decoding for Efficient LLM Test-Time Scaling","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-05T13:48:21.885995Z"},"links":{"cited_paper":"/paper/2305.20050","citing_paper":"/paper/2509.04474"},"observation_digest":"sha256:89ecb3fc2012ce2ee303bb0be635c7d960ca628e7c7c7748516a5d19596619a2","observation_id":"e2392494-a000-412a-a0a1-822dcb38bf9e","resolution":{"observed_at":"2026-08-05T13:48:21.885995Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2505.17155","last_updated":"2025-05-31T13:54:42Z","snapshot_observed_at":"2026-08-13T19:59:22.724751Z","submitted_at":"2025-05-22T12:23:30Z","title":"TrimR: Verifier-based Training-Free Thinking Compression for Efficient Test-Time Scaling","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.17155","snapshot_observed_at":"2026-08-05T13:48:21.999552Z","title":"Cmcts: A constrained monte carlo tree search framework for mathematical reasoning in large language model, 2025a","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2509.04474","last_updated":"2025-08-30T01:54:55Z","snapshot_observed_at":"2026-08-17T14:22:16.859857Z","submitted_at":"2025-08-30T01:54:55Z","title":"Scaling Up, Speeding Up: A Benchmark of Speculative Decoding for Efficient LLM Test-Time Scaling","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-05T13:48:21.999552Z"},"links":{"cited_paper":"/paper/2505.17155","citing_paper":"/paper/2509.04474"},"observation_digest":"sha256:5edd25eaba6dbe0f66ff3c5ad25c92ccdc0f362b25db52ea7f9913bc7600b583","observation_id":"b75ec70c-a409-46f9-af64-e1c149f1ad54","resolution":{"observed_at":"2026-08-05T13:48:21.999552Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2504.09858","last_updated":"2025-04-14T04:08:16Z","snapshot_observed_at":"2026-08-16T12:41:30.521725Z","submitted_at":"2025-04-14T04:08:16Z","title":"Reasoning Models Can Be Effective Without Thinking","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.09858","snapshot_observed_at":"2026-08-05T13:48:22.081867Z","title":"Reasoning models can be effective without thinking","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2509.04474","last_updated":"2025-08-30T01:54:55Z","snapshot_observed_at":"2026-08-17T14:22:16.859857Z","submitted_at":"2025-08-30T01:54:55Z","title":"Scaling Up, Speeding Up: A Benchmark of Speculative Decoding for Efficient LLM Test-Time Scaling","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-05T13:48:22.081867Z"},"links":{"cited_paper":"/paper/2504.09858","citing_paper":"/paper/2509.04474"},"observation_digest":"sha256:fd1c0a5358007de0ffe169d7c2bf57bb4d2cbca2ba163f01ad4c5029c168f492","observation_id":"5498e677-3537-4449-a6be-a819a8f98388","resolution":{"observed_at":"2026-08-05T13:48:22.081867Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T13:48:22.195104Z","title":"Suffixdecoding: Extreme speculative decoding for emerging ai applications","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2509.04474","last_updated":"2025-08-30T01:54:55Z","snapshot_observed_at":"2026-08-17T14:22:16.859857Z","submitted_at":"2025-08-30T01:54:55Z","title":"Scaling Up, Speeding Up: A Benchmark of Speculative Decoding for Efficient LLM Test-Time Scaling","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-05T13:48:22.195104Z"},"links":{"citing_paper":"/paper/2509.04474"},"observation_digest":"sha256:10dc26d35affeefd7c7ca71d6f88ef57d73248541eeed67e7c6056df4d27c6e4","observation_id":"652600af-233b-4e66-b3b0-db8b68ae23de","resolution":{"observed_at":"2026-08-05T13:48:22.195104Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.16720","last_updated":"2026-04-30T02:46:40Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-12-21T18:04:31Z","title":"OpenAI o1 System Card","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.16720","snapshot_observed_at":"2026-08-05T13:48:22.337886Z","title":"Openai o1 system card","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2509.04474","last_updated":"2025-08-30T01:54:55Z","snapshot_observed_at":"2026-08-17T14:22:16.859857Z","submitted_at":"2025-08-30T01:54:55Z","title":"Scaling Up, Speeding Up: A Benchmark of Speculative Decoding for Efficient LLM Test-Time Scaling","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-05T13:48:22.337886Z"},"links":{"cited_paper":"/paper/2412.16720","citing_paper":"/paper/2509.04474"},"observation_digest":"sha256:5f9beddd86c305596f23ba784495e8872906791d2af6bd00d668ba3bd6a0fe0d","observation_id":"e740dd57-f9fe-439c-b562-be754c3a9e5e","resolution":{"observed_at":"2026-08-05T13:48:22.337886Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2504.07891","last_updated":"2025-05-16T19:27:32Z","snapshot_observed_at":"2026-08-16T16:44:51.424694Z","submitted_at":"2025-04-10T16:05:19Z","title":"SpecReason: Fast and Accurate Inference-Time Compute via Speculative Reasoning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.07891","snapshot_observed_at":"2026-08-05T13:48:22.435299Z","title":"Specreason: Fast and accurate inference-time compute via speculative reasoning","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2509.04474","last_updated":"2025-08-30T01:54:55Z","snapshot_observed_at":"2026-08-17T14:22:16.859857Z","submitted_at":"2025-08-30T01:54:55Z","title":"Scaling Up, Speeding Up: A Benchmark of Speculative Decoding for Efficient LLM Test-Time Scaling","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-05T13:48:22.435299Z"},"links":{"cited_paper":"/paper/2504.07891","citing_paper":"/paper/2509.04474"},"observation_digest":"sha256:3347321c9bbf85a4a3394ebd8d6d2a4b2d5d78d905542816c0a9ab26b9d05a96","observation_id":"5774eab7-57c0-4983-a394-583f0c9d3f03","resolution":{"observed_at":"2026-08-05T13:48:22.435299Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.12022","last_updated":"2023-11-20T18:57:34Z","snapshot_observed_at":"2026-08-16T07:06:07.822225Z","submitted_at":"2023-11-20T18:57:34Z","title":"GPQA: A Graduate-Level Google-Proof Q&A Benchmark","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.12022","snapshot_observed_at":"2026-08-05T13:48:22.547651Z","title":"Apoorv Saxena","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2509.04474","last_updated":"2025-08-30T01:54:55Z","snapshot_observed_at":"2026-08-17T14:22:16.859857Z","submitted_at":"2025-08-30T01:54:55Z","title":"Scaling Up, Speeding Up: A Benchmark of Speculative Decoding for Efficient LLM Test-Time Scaling","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-05T13:48:22.547651Z"},"links":{"cited_paper":"/paper/2311.12022","citing_paper":"/paper/2509.04474"},"observation_digest":"sha256:70d7ae3a77af92ce33d01bff41f5963248d1a78269eb70f269af855ea674e328","observation_id":"d1a4389e-4c0a-4ff9-a3e4-15cf770075bc","resolution":{"observed_at":"2026-08-05T13:48:22.547651Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.16419","last_updated":"2025-08-21T19:14:40Z","snapshot_observed_at":"2026-08-11T13:10:23.709172Z","submitted_at":"2025-03-20T17:59:38Z","title":"Stop Overthinking: A Survey on Efficient Reasoning for Large Language Models","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.16419","snapshot_observed_at":"2026-08-05T13:48:22.659104Z","title":"Stop overthinking: A survey on efficient reasoning for large language models","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2509.04474","last_updated":"2025-08-30T01:54:55Z","snapshot_observed_at":"2026-08-17T14:22:16.859857Z","submitted_at":"2025-08-30T01:54:55Z","title":"Scaling Up, Speeding Up: A Benchmark of Speculative Decoding for Efficient LLM Test-Time Scaling","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-05T13:48:22.659104Z"},"links":{"cited_paper":"/paper/2503.16419","citing_paper":"/paper/2509.04474"},"observation_digest":"sha256:f3fbf723a3800a0372b332e26470dda084f8be02b133f21c57c37ab87730bd87","observation_id":"fc50e29f-9b0c-4dff-9163-54b06bb2b2c0","resolution":{"observed_at":"2026-08-05T13:48:22.659104Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.19855","last_updated":"2025-03-25T17:19:38Z","snapshot_observed_at":"2026-08-16T12:46:54.354556Z","submitted_at":"2025-03-25T17:19:38Z","title":"Think Twice: Enhancing LLM Reasoning by Scaling Multi-round Test-time Thinking","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.19855","snapshot_observed_at":"2026-08-05T13:48:22.756427Z","title":"Think twice: Enhancing LLM reasoning by scaling multi-round test-time thinking","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2509.04474","last_updated":"2025-08-30T01:54:55Z","snapshot_observed_at":"2026-08-17T14:22:16.859857Z","submitted_at":"2025-08-30T01:54:55Z","title":"Scaling Up, Speeding Up: A Benchmark of Speculative Decoding for Efficient LLM Test-Time Scaling","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-05T13:48:22.756427Z"},"links":{"cited_paper":"/paper/2503.19855","citing_paper":"/paper/2509.04474"},"observation_digest":"sha256:de83304deb0d062a37bf5126e1982ca865d10486aab03e370ac39f4882d9a721","observation_id":"248c6c22-6920-42e5-9b6a-c6d3a7e41172","resolution":{"observed_at":"2026-08-05T13:48:22.756427Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T13:48:22.822091Z","title":"R1-compress: Long chain-of-thought compression via chunk compres- sion and search","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2509.04474","last_updated":"2025-08-30T01:54:55Z","snapshot_observed_at":"2026-08-17T14:22:16.859857Z","submitted_at":"2025-08-30T01:54:55Z","title":"Scaling Up, Speeding Up: A Benchmark of Speculative Decoding for Efficient LLM Test-Time Scaling","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-05T13:48:22.822091Z"},"links":{"citing_paper":"/paper/2509.04474"},"observation_digest":"sha256:44d1aa53e89726e2bdfd5d893bdc187e158ce58847c115d0c8c2cc31bac0fec2","observation_id":"66927755-e89d-481b-97db-e4e91bf8aeeb","resolution":{"observed_at":"2026-08-05T13:48:22.822091Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.20641","last_updated":"2025-05-23T02:41:43Z","snapshot_observed_at":"2026-08-16T23:38:30.883715Z","submitted_at":"2025-03-26T15:34:37Z","title":"Unlocking Efficient Long-to-Short LLM Reasoning with Model Merging","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.20641","snapshot_observed_at":"2026-08-05T13:48:22.954670Z","title":"Unlocking efficient long-to-short llm reasoning with model merging","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2509.04474","last_updated":"2025-08-30T01:54:55Z","snapshot_observed_at":"2026-08-17T14:22:16.859857Z","submitted_at":"2025-08-30T01:54:55Z","title":"Scaling Up, Speeding Up: A Benchmark of Speculative Decoding for Efficient LLM Test-Time Scaling","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-05T13:48:22.954670Z"},"links":{"cited_paper":"/paper/2503.20641","citing_paper":"/paper/2509.04474"},"observation_digest":"sha256:255da555262f2c9b3ba36865277f06fcfc5c0347f06b9853e7a7d6df5bff8da8","observation_id":"178cae99-eb4f-45a1-9c87-f08a431ad679","resolution":{"observed_at":"2026-08-05T13:48:22.954670Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2505.09388","last_updated":"2025-05-14T13:41:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-14T13:41:34Z","title":"Qwen3 Technical Report","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.09388","snapshot_observed_at":"2026-08-05T13:48:23.073663Z","title":"Qwen3 techni- cal report","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2509.04474","last_updated":"2025-08-30T01:54:55Z","snapshot_observed_at":"2026-08-17T14:22:16.859857Z","submitted_at":"2025-08-30T01:54:55Z","title":"Scaling Up, Speeding Up: A Benchmark of Speculative Decoding for Efficient LLM Test-Time Scaling","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-05T13:48:23.073663Z"},"links":{"cited_paper":"/paper/2505.09388","citing_paper":"/paper/2509.04474"},"observation_digest":"sha256:770eaf1596428dfe758f4bf7c82aae329205c9e2626e878e2db4449d6de9f7e1","observation_id":"d24ca8c5-ad1c-4519-b0ba-d3b3ceea96cf","resolution":{"observed_at":"2026-08-05T13:48:23.073663Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.24235","last_updated":"2025-05-04T15:48:08Z","snapshot_observed_at":"2026-08-16T01:22:08.293337Z","submitted_at":"2025-03-31T15:46:15Z","title":"A Survey on Test-Time Scaling in Large Language Models: What, How, Where, and How Well?","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.24235","snapshot_observed_at":"2026-08-05T13:48:23.144764Z","title":"A survey on test-time scaling in large language models: What, how, where, and how well? In arXiv:2503.24235,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2509.04474","last_updated":"2025-08-30T01:54:55Z","snapshot_observed_at":"2026-08-17T14:22:16.859857Z","submitted_at":"2025-08-30T01:54:55Z","title":"Scaling Up, Speeding Up: A Benchmark of Speculative Decoding for Efficient LLM Test-Time Scaling","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-05T13:48:23.144764Z"},"links":{"cited_paper":"/paper/2503.24235","citing_paper":"/paper/2509.04474"},"observation_digest":"sha256:6f324f3fb8ec7b298a98bf4f79d874a68f6cc28b90b0a8bf280ee8643cd91c23","observation_id":"5ff7f1ea-4082-41c1-90fb-2770e16904bf","resolution":{"observed_at":"2026-08-05T13:48:23.144764Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12948","last_updated":"2026-01-04T03:57:36Z","snapshot_observed_at":"2026-08-15T12:33:55.451951Z","submitted_at":"2025-01-22T15:19:35Z","title":"DeepSeek-R1: Incentivizing Reasoning Capability in LLMs via Reinforcement Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.12948","snapshot_observed_at":"2026-08-05T13:48:21.341874Z","title":"Deepseek-r1: Incentivizing reasoning capability in llms via reinforcement learning","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2509.04474","last_updated":"2025-08-30T01:54:55Z","snapshot_observed_at":"2026-08-17T14:22:16.859857Z","submitted_at":"2025-08-30T01:54:55Z","title":"Scaling Up, Speeding Up: A Benchmark of Speculative Decoding for Efficient LLM Test-Time Scaling","version":1},"reference_index":2021,"source":"pdf_text","source_observed_at":"2026-08-05T13:48:21.341874Z"},"links":{"cited_paper":"/paper/2501.12948","citing_paper":"/paper/2509.04474"},"observation_digest":"sha256:0fbdadad09d5f3f210ef30778d21e038a702fade5392592cee84dbd2e4717a3a","observation_id":"393a2bd1-c2b2-4eab-a223-8903dcaf506e","resolution":{"observed_at":"2026-08-05T13:48:21.341874Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2505.14183","last_updated":"2025-05-20T10:40:41Z","snapshot_observed_at":"2026-08-17T12:21:54.085062Z","submitted_at":"2025-05-20T10:40:41Z","title":"ThinkSwitcher: When to Think Hard, When to Think Fast","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.14183","snapshot_observed_at":"2026-08-05T13:48:21.764917Z","title":"Thinkswitcher: When to think hard, when to think fast","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2509.04474","last_updated":"2025-08-30T01:54:55Z","snapshot_observed_at":"2026-08-17T14:22:16.859857Z","submitted_at":"2025-08-30T01:54:55Z","title":"Scaling Up, Speeding Up: A Benchmark of Speculative Decoding for Efficient LLM Test-Time Scaling","version":1},"reference_index":2022,"source":"pdf_text","source_observed_at":"2026-08-05T13:48:21.764917Z"},"links":{"cited_paper":"/paper/2505.14183","citing_paper":"/paper/2509.04474"},"observation_digest":"sha256:f82da3a8c6bbb15a7ed39f8f2073258a703f6275034bc9097e0a919b6bdc2a09","observation_id":"591a950f-0fcf-4023-b754-e1aa1b44acc3","resolution":{"observed_at":"2026-08-05T13:48:21.764917Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2110.14168","last_updated":"2021-11-18T00:23:45Z","snapshot_observed_at":"2026-08-14T02:43:01.480086Z","submitted_at":"2021-10-27T04:49:45Z","title":"Training Verifiers to Solve Math Word Problems","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2110.14168","snapshot_observed_at":"2026-08-05T13:48:21.248460Z","title":"Training verifiers to solve math word problems","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2509.04474","last_updated":"2025-08-30T01:54:55Z","snapshot_observed_at":"2026-08-17T14:22:16.859857Z","submitted_at":"2025-08-30T01:54:55Z","title":"Scaling Up, Speeding Up: A Benchmark of Speculative Decoding for Efficient LLM Test-Time Scaling","version":1},"reference_index":2023,"source":"pdf_text","source_observed_at":"2026-08-05T13:48:21.248460Z"},"links":{"cited_paper":"/paper/2110.14168","citing_paper":"/paper/2509.04474"},"observation_digest":"sha256:e426ba08691b345d25b29a1895efe4bb4de1952c8d953f09002599dc7e9960d5","observation_id":"eb14ed75-474b-4cd7-87a3-cce4ce4bebb3","resolution":{"observed_at":"2026-08-05T13:48:21.248460Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.10774","last_updated":"2024-06-14T23:32:32Z","snapshot_observed_at":"2026-08-17T10:13:54.763177Z","submitted_at":"2024-01-19T15:48:40Z","title":"Medusa: Simple LLM Inference Acceleration Framework with Multiple Decoding Heads","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.10774","snapshot_observed_at":"2026-08-05T13:48:21.101641Z","title":"Lee, Deming Chen, and Tri Dao","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2509.04474","last_updated":"2025-08-30T01:54:55Z","snapshot_observed_at":"2026-08-17T14:22:16.859857Z","submitted_at":"2025-08-30T01:54:55Z","title":"Scaling Up, Speeding Up: A Benchmark of Speculative Decoding for Efficient LLM Test-Time Scaling","version":1},"reference_index":2024,"source":"pdf_text","source_observed_at":"2026-08-05T13:48:21.101641Z"},"links":{"cited_paper":"/paper/2401.10774","citing_paper":"/paper/2509.04474"},"observation_digest":"sha256:5fd516c81490d660d2c66bf507d01c0156f425ac9bcb3e5482beb7fd3c4cd5d8","observation_id":"ef734227-f47c-4a37-9275-0c36f298f419","resolution":{"observed_at":"2026-08-05T13:48:21.101641Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.02057","last_updated":"2024-02-03T06:37:50Z","snapshot_observed_at":"2026-08-16T14:21:54.969828Z","submitted_at":"2024-02-03T06:37:50Z","title":"Break the Sequential Dependency of LLM Inference Using Lookahead Decoding","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.02057","snapshot_observed_at":"2026-08-05T13:48:21.420932Z","title":"Break the sequential dependency of llm infer- ence using lookahead decoding","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2509.04474","last_updated":"2025-08-30T01:54:55Z","snapshot_observed_at":"2026-08-17T14:22:16.859857Z","submitted_at":"2025-08-30T01:54:55Z","title":"Scaling Up, Speeding Up: A Benchmark of Speculative Decoding for Efficient LLM Test-Time Scaling","version":1},"reference_index":2025,"source":"pdf_text","source_observed_at":"2026-08-05T13:48:21.420932Z"},"links":{"cited_paper":"/paper/2402.02057","citing_paper":"/paper/2509.04474"},"observation_digest":"sha256:bc55a8729676b50e557861e0555fdcb0d2d1d0e534f480f06291902c09e733ed","observation_id":"c386ed0c-2dca-44de-98e1-0ab2d484f2ae","resolution":{"observed_at":"2026-08-05T13:48:21.420932Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2509.04474","last_updated":"2025-08-30T01:54:55Z","latest_version":1,"primary_category":"cs.CL","snapshot_observed_at":"2026-08-17T14:22:16.859857Z","submitted_at":"2025-08-30T01:54:55Z","title":"Scaling Up, Speeding Up: A Benchmark of Speculative Decoding for Efficient LLM Test-Time Scaling"},"reference_resolution":{"displayed":22,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":22,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":22},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"thesis":"As of 17 August 2026, this Paper Citation Record lists 22 of 22 outbound references and 0 inbound Pith citation observations for arXiv:2509.04474."}