{"as_of":"2026-08-18T22:02:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:3d8ac9a20423b749513733e2ee5631287bfdbe9a8e516413df887c1ec7bf254d","coverage":[{"denominator":40,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":40,"source":"paper_references, paper_reference_links","source_observed_at":"2026-05-20T02:11:26.925234Z","state":"measured"},{"denominator":42,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":42,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-18T06:34:40.430872+00:00","state":"measured"},{"denominator":2,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":2,"source":"paper_references, paper_reference_links","source_observed_at":"2026-07-11T01:55:09.658053Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"pith","source_observed_at":"2026-07-11T01:57:51.122520Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2605.19775","last_updated":"2026-05-19T12:43:51Z","snapshot_observed_at":"2026-08-08T23:12:58.922286Z","submitted_at":"2026-05-19T12:43:51Z","title":"Understanding Inference Scaling for LLMs: Bottlenecks, Trade-offs, and Performance Principles","version":1},"cited_work":{"arxiv_id":"2605.19775","doi":null,"metadata_source":"pith","pith_arxiv_id":"2605.19775","snapshot_observed_at":"2026-07-11T01:57:51.122520Z","title":"Understanding Inference Scaling for LLMs: Bottlenecks, Trade-offs, and Performance Principles","venue":"cs.DC","work_id":"9dd1ab5a-2cb8-44e6-b910-41a53dba4d25","year":2026},"citing_paper":{"arxiv_id":"2607.05876","last_updated":"2026-07-08T02:47:41Z","snapshot_observed_at":"2026-08-05T07:46:13.093303Z","submitted_at":"2026-07-07T06:11:54Z","title":"Think Before You Grid-Search: Floor-First Triage for LLM Serving","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-07-08T22:38:12.637901Z"},"links":{"cited_paper":"/paper/2605.19775","citing_paper":"/paper/2607.05876"},"observation_digest":"sha256:19b6a44627791cebae03c04928d1c1cb118bb34055f96e77af78b08cf5f84599","observation_id":"e16fca69-1acf-403f-88c4-af13a4444c3a","resolution":{"observed_at":"2026-07-08T22:45:40.093424Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2605.19775","last_updated":"2026-05-19T12:43:51Z","snapshot_observed_at":"2026-08-08T23:12:58.922286Z","submitted_at":"2026-05-19T12:43:51Z","title":"Understanding Inference Scaling for LLMs: Bottlenecks, Trade-offs, and Performance Principles","version":1},"cited_work":{"arxiv_id":"2605.19775","doi":null,"metadata_source":"pith","pith_arxiv_id":"2605.19775","snapshot_observed_at":"2026-07-11T01:57:51.122520Z","title":"Understanding Inference Scaling for LLMs: Bottlenecks, Trade-offs, and Performance Principles","venue":"cs.DC","work_id":"9dd1ab5a-2cb8-44e6-b910-41a53dba4d25","year":2026},"citing_paper":{"arxiv_id":"2607.05876","last_updated":"2026-07-08T02:47:41Z","snapshot_observed_at":"2026-08-05T07:46:13.093303Z","submitted_at":"2026-07-07T06:11:54Z","title":"Think Before You Grid-Search: Floor-First Triage for LLM Serving","version":2},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-07-11T01:55:09.658053Z"},"links":{"cited_paper":"/paper/2605.19775","citing_paper":"/paper/2607.05876"},"observation_digest":"sha256:eb0c4eaeb73b82441f04d34350ee7ddfa308048a717cfd6dbf2b28318e3d47b8","observation_id":"4c2983f9-187a-4a27-8fc3-bc6bf4e8b52e","resolution":{"observed_at":"2026-07-11T01:57:51.154963Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2605.19775/citation-record","integrity":"/paper/2605.19775/integrity","json":"/paper/2605.19775/citation-record.json","paper":"/paper/2605.19775"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Vidur: A large-scale simulation frame- work for llm inference","venue":null,"work_id":"db446baa-ff97-459c-9efb-d4d9829ad1ea","year":2024},"citing_paper":{"arxiv_id":"2605.19775","last_updated":"2026-05-19T12:43:51Z","snapshot_observed_at":"2026-08-08T23:12:58.922286Z","submitted_at":"2026-05-19T12:43:51Z","title":"Understanding Inference Scaling for LLMs: Bottlenecks, Trade-offs, and Performance Principles","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-20T02:11:26.925234Z"},"links":{"citing_paper":"/paper/2605.19775"},"observation_digest":"sha256:24f906e0d47bb814b8e756f1041e048c4511a2fb10fdd825527b928672ce638a","observation_id":"cc66ec53-c92b-4c92-ac79-50a1fe7f11c5","resolution":{"observed_at":"2026-05-20T02:12:58.713366Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Taming{Throughput-Latency}tradeoff in{LLM}inference with{Sarathi-Serve}","venue":null,"work_id":"8cbbafdc-5c7f-457e-912d-b7c55e062ca5","year":2024},"citing_paper":{"arxiv_id":"2605.19775","last_updated":"2026-05-19T12:43:51Z","snapshot_observed_at":"2026-08-08T23:12:58.922286Z","submitted_at":"2026-05-19T12:43:51Z","title":"Understanding Inference Scaling for LLMs: Bottlenecks, Trade-offs, and Performance Principles","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-20T02:11:26.925234Z"},"links":{"citing_paper":"/paper/2605.19775"},"observation_digest":"sha256:fdca1cd8f0e8e8408f360f7500061f42e2fa9fa91b3cc26faf5c89fc84552486","observation_id":"38c70ef2-8da4-4000-9549-73029ad015d9","resolution":{"observed_at":"2026-05-20T02:12:58.790233Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.16369","last_updated":"2023-08-31T00:03:02Z","snapshot_observed_at":"2026-08-15T04:42:28.752204Z","submitted_at":"2023-08-31T00:03:02Z","title":"SARATHI: Efficient LLM Inference by Piggybacking Decodes with Chunked Prefills","version":1},"cited_work":{"arxiv_id":"2308.16369","doi":"10.48550/arxiv.2308.16369","metadata_source":"pith","pith_arxiv_id":"2308.16369","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SARATHI: Efficient LLM Inference by Piggybacking Decodes with Chunked Prefills","venue":"cs.LG","work_id":"3dbdd757-ca01-436f-acfd-12ffcd6f64c6","year":2023},"citing_paper":{"arxiv_id":"2605.19775","last_updated":"2026-05-19T12:43:51Z","snapshot_observed_at":"2026-08-08T23:12:58.922286Z","submitted_at":"2026-05-19T12:43:51Z","title":"Understanding Inference Scaling for LLMs: Bottlenecks, Trade-offs, and Performance Principles","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-20T02:11:26.925234Z"},"links":{"cited_paper":"/paper/2308.16369","citing_paper":"/paper/2605.19775"},"observation_digest":"sha256:ab72154bd917d4db89f02aa8b796452df0aff81a90664cf0e2e617ddf94efd2c","observation_id":"2dd0a23f-066c-4141-9f14-2f9b8061721e","resolution":{"observed_at":"2026-05-20T02:12:58.271581Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.13245","last_updated":"2023-12-23T17:55:11Z","snapshot_observed_at":"2026-08-15T06:28:09.529747Z","submitted_at":"2023-05-22T17:16:38Z","title":"GQA: Training Generalized Multi-Query Transformer Models from Multi-Head Checkpoints","version":3},"cited_work":{"arxiv_id":"2305.13245","doi":"10.48550/arxiv.2305.13245","metadata_source":"pith","pith_arxiv_id":"2305.13245","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"GQA: Training Generalized Multi-Query Transformer Models from Multi-Head Checkpoints","venue":"cs.CL","work_id":"b73ad5b2-e553-4c71-b0c9-67e67ba7b158","year":2023},"citing_paper":{"arxiv_id":"2605.19775","last_updated":"2026-05-19T12:43:51Z","snapshot_observed_at":"2026-08-08T23:12:58.922286Z","submitted_at":"2026-05-19T12:43:51Z","title":"Understanding Inference Scaling for LLMs: Bottlenecks, Trade-offs, and Performance Principles","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-20T02:11:26.925234Z"},"links":{"cited_paper":"/paper/2305.13245","citing_paper":"/paper/2605.19775"},"observation_digest":"sha256:b09b64728ffb1eed71e17b3fce6ac43e0de1fb21b279732ed18c04ac63121667","observation_id":"0f45074d-80ae-42d8-aad2-6c39e26f2718","resolution":{"observed_at":"2026-05-20T02:12:58.293100Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Llm in a flash: Efficient large language model inference with limited memory","venue":null,"work_id":"6ee70cd3-d3e7-46c0-9b0b-45e5dac8737c","year":2024},"citing_paper":{"arxiv_id":"2605.19775","last_updated":"2026-05-19T12:43:51Z","snapshot_observed_at":"2026-08-08T23:12:58.922286Z","submitted_at":"2026-05-19T12:43:51Z","title":"Understanding Inference Scaling for LLMs: Bottlenecks, Trade-offs, and Performance Principles","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-05-20T02:11:26.925234Z"},"links":{"citing_paper":"/paper/2605.19775"},"observation_digest":"sha256:7c573f205e6ad364cbbc44e2bbe1d023c0219c319ddb3485cfafc98650a340b7","observation_id":"854ab92f-9190-4b23-90fb-8d4efb86f63c","resolution":{"observed_at":"2026-05-20T02:12:58.717522Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"5008.354505","doi":"10.1145/3545008.3545054","metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Exploiting cxl-based memory for distributed deep learning","venue":null,"work_id":"1516272a-eba5-4734-bb41-5d5e526ed2ab","year":2023},"citing_paper":{"arxiv_id":"2605.19775","last_updated":"2026-05-19T12:43:51Z","snapshot_observed_at":"2026-08-08T23:12:58.922286Z","submitted_at":"2026-05-19T12:43:51Z","title":"Understanding Inference Scaling for LLMs: Bottlenecks, Trade-offs, and Performance Principles","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-05-20T02:11:26.925234Z"},"links":{"citing_paper":"/paper/2605.19775"},"observation_digest":"sha256:479d7e940451f61d30bfe25e9fa358d301e0bc5c567b004ee42d568696316a1b","observation_id":"0092719e-4193-4989-9453-d5ad9dacc9cc","resolution":{"observed_at":"2026-05-20T02:12:57.475972Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-05-20T07:54:09.599174+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-20T07:54:09.599174+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"9013.359667","doi":"10.1145/3589013.3596678","metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Accelerating performance of gpu-based workloads using cxl","venue":null,"work_id":"c8c405aa-7c11-4f2e-a292-3ca4b37ccb8e","year":2023},"citing_paper":{"arxiv_id":"2605.19775","last_updated":"2026-05-19T12:43:51Z","snapshot_observed_at":"2026-08-08T23:12:58.922286Z","submitted_at":"2026-05-19T12:43:51Z","title":"Understanding Inference Scaling for LLMs: Bottlenecks, Trade-offs, and Performance Principles","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-05-20T02:11:26.925234Z"},"links":{"citing_paper":"/paper/2605.19775"},"observation_digest":"sha256:a3bebc009b87368bb43bf829a7270c12a1afa6f7b2c855e0bb7b23e3b37424c6","observation_id":"cb609ecb-212b-4d90-8ee7-f09461b3bce5","resolution":{"observed_at":"2026-05-20T02:12:57.489835Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-05-20T07:54:10.02119+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-20T07:54:10.02119+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Moe-lightning: High-throughput moe inference on memory-constrained gpus","venue":null,"work_id":"63e3da2a-4e68-496e-9ccf-675ce6233f97","year":2025},"citing_paper":{"arxiv_id":"2605.19775","last_updated":"2026-05-19T12:43:51Z","snapshot_observed_at":"2026-08-08T23:12:58.922286Z","submitted_at":"2026-05-19T12:43:51Z","title":"Understanding Inference Scaling for LLMs: Bottlenecks, Trade-offs, and Performance Principles","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-05-20T02:11:26.925234Z"},"links":{"citing_paper":"/paper/2605.19775"},"observation_digest":"sha256:0e71af014901d2960b00d97bebfa2b69593eb1ecbd543dfa7311fa8dca2e4f2f","observation_id":"2847a015-37f3-497a-93af-7dff61efb162","resolution":{"observed_at":"2026-05-20T02:12:58.780957Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2510.09665","doi":"10.48550/arxiv.2510.09665","metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Lmcache: An efficient kv cache layer for enterprise-scale llm inference","venue":"arXiv (Cornell University)","work_id":"089b937e-f3f9-4525-a792-524ef0f7db1d","year":2025},"citing_paper":{"arxiv_id":"2605.19775","last_updated":"2026-05-19T12:43:51Z","snapshot_observed_at":"2026-08-08T23:12:58.922286Z","submitted_at":"2026-05-19T12:43:51Z","title":"Understanding Inference Scaling for LLMs: Bottlenecks, Trade-offs, and Performance Principles","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-05-20T02:11:26.925234Z"},"links":{"citing_paper":"/paper/2605.19775"},"observation_digest":"sha256:f67524fef14cfde75543513f08273e51b3271ccba5ff9b954b2ae7d593613ae1","observation_id":"c290b306-54b3-42df-be41-c345b222bbce","resolution":{"observed_at":"2026-05-20T02:12:58.246024Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Llm-inference- bench: Inference benchmarking of large language models on ai acceler- ators","venue":null,"work_id":"1673ff30-261d-4a3e-ae29-873742424bcb","year":2024},"citing_paper":{"arxiv_id":"2605.19775","last_updated":"2026-05-19T12:43:51Z","snapshot_observed_at":"2026-08-08T23:12:58.922286Z","submitted_at":"2026-05-19T12:43:51Z","title":"Understanding Inference Scaling for LLMs: Bottlenecks, Trade-offs, and Performance Principles","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-05-20T02:11:26.925234Z"},"links":{"citing_paper":"/paper/2605.19775"},"observation_digest":"sha256:0d7e5f91d4297895c09c31c6b480396a1bcd2f9bfc415654610d410b6842ce54","observation_id":"f109b5d0-54b4-4d5d-bbdc-be32baacb865","resolution":{"observed_at":"2026-05-20T02:12:58.721706Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"PagedEviction: Structured block-wise KV cache pruning for efficient large language model inference","venue":null,"work_id":"9a8bdc57-fd91-48aa-9f40-a360862079ed","year":2026},"citing_paper":{"arxiv_id":"2605.19775","last_updated":"2026-05-19T12:43:51Z","snapshot_observed_at":"2026-08-08T23:12:58.922286Z","submitted_at":"2026-05-19T12:43:51Z","title":"Understanding Inference Scaling for LLMs: Bottlenecks, Trade-offs, and Performance Principles","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-05-20T02:11:26.925234Z"},"links":{"citing_paper":"/paper/2605.19775"},"observation_digest":"sha256:cba918ee7173d8dc65bff1d16945178c83cbced167086e8dfb3659869b2dbfac","observation_id":"589e859c-81ef-40aa-8b51-2161b16c942d","resolution":{"observed_at":"2026-05-20T02:12:58.771929Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2006.16362","last_updated":"2021-05-20T14:48:30Z","snapshot_observed_at":"2026-08-09T06:13:20.021836Z","submitted_at":"2020-06-29T20:28:52Z","title":"Multi-Head Attention: Collaborate Instead of Concatenate","version":2},"cited_work":{"arxiv_id":"2006.16362","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2006.16362","snapshot_observed_at":"2026-07-03T23:59:07.199652Z","title":"Multi-head attention: Collaborate instead of concatenate","venue":null,"work_id":"122d504b-cee2-47b2-afb8-f2bad267791e","year":2006},"citing_paper":{"arxiv_id":"2605.19775","last_updated":"2026-05-19T12:43:51Z","snapshot_observed_at":"2026-08-08T23:12:58.922286Z","submitted_at":"2026-05-19T12:43:51Z","title":"Understanding Inference Scaling for LLMs: Bottlenecks, Trade-offs, and Performance Principles","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-05-20T02:11:26.925234Z"},"links":{"cited_paper":"/paper/2006.16362","citing_paper":"/paper/2605.19775"},"observation_digest":"sha256:aabe901d45ce73e97b0e0c6c0e6baef32927e46a7467f36929b95d65ebda6783","observation_id":"2a9a8dd0-c925-4770-b78a-d4aefb86c975","resolution":{"observed_at":"2026-05-20T02:12:58.240864Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Compute express link","venue":null,"work_id":"bb1143d5-dc95-44cf-8415-9d57be13184d","year":2025},"citing_paper":{"arxiv_id":"2605.19775","last_updated":"2026-05-19T12:43:51Z","snapshot_observed_at":"2026-08-08T23:12:58.922286Z","submitted_at":"2026-05-19T12:43:51Z","title":"Understanding Inference Scaling for LLMs: Bottlenecks, Trade-offs, and Performance Principles","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-05-20T02:11:26.925234Z"},"links":{"citing_paper":"/paper/2605.19775"},"observation_digest":"sha256:b6b30dbddf56713aac8e8877c90c5c3940ea5f1f10ad0cf4aaaf594cd9f7e813","observation_id":"f9307ee1-4d2d-4b79-8bff-1e65af4f311d","resolution":{"observed_at":"2026-05-20T02:12:58.725450Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Corsair™. built for generative a","venue":null,"work_id":"c69f6284-987c-4129-bade-7c0fda337f61","year":2025},"citing_paper":{"arxiv_id":"2605.19775","last_updated":"2026-05-19T12:43:51Z","snapshot_observed_at":"2026-08-08T23:12:58.922286Z","submitted_at":"2026-05-19T12:43:51Z","title":"Understanding Inference Scaling for LLMs: Bottlenecks, Trade-offs, and Performance Principles","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-05-20T02:11:26.925234Z"},"links":{"citing_paper":"/paper/2605.19775"},"observation_digest":"sha256:216f04b1de9c68d5cfdfedbc96b04c3768ad8771916b6db87ce75905443ae713","observation_id":"6d6d5136-e4d2-4ce1-941c-7d307fca72c2","resolution":{"observed_at":"2026-05-20T02:12:58.768204Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Available: https://www.d-matrix.ai/product/","venue":null,"work_id":"242da7a3-b609-48ef-b898-fae3771d377f","year":null},"citing_paper":{"arxiv_id":"2605.19775","last_updated":"2026-05-19T12:43:51Z","snapshot_observed_at":"2026-08-08T23:12:58.922286Z","submitted_at":"2026-05-19T12:43:51Z","title":"Understanding Inference Scaling for LLMs: Bottlenecks, Trade-offs, and Performance Principles","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-05-20T02:11:26.925234Z"},"links":{"citing_paper":"/paper/2605.19775"},"observation_digest":"sha256:d8047b8033199bd73f5c9d8ff67a92ce2e74594268a2f421a7e18c3139a5530a","observation_id":"caa864e8-407d-4b4c-9f34-8013f04aa19c","resolution":{"observed_at":"2026-05-20T02:12:58.743392Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Why we decoupled execution to accelerate i/o","venue":null,"work_id":"f32990da-4d57-40de-92d5-ef04c7e61422","year":2025},"citing_paper":{"arxiv_id":"2605.19775","last_updated":"2026-05-19T12:43:51Z","snapshot_observed_at":"2026-08-08T23:12:58.922286Z","submitted_at":"2026-05-19T12:43:51Z","title":"Understanding Inference Scaling for LLMs: Bottlenecks, Trade-offs, and Performance Principles","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-05-20T02:11:26.925234Z"},"links":{"citing_paper":"/paper/2605.19775"},"observation_digest":"sha256:6bf704ca675a1a08c2db120a64ebde1d40e94cd6b78402824c8655260fb81aa2","observation_id":"b9e4e9d7-a603-4429-9d15-7a292149cec6","resolution":{"observed_at":"2026-05-20T02:12:58.755588Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.12948","last_updated":"2026-01-04T03:57:36Z","snapshot_observed_at":"2026-08-15T12:33:55.451951Z","submitted_at":"2025-01-22T15:19:35Z","title":"DeepSeek-R1: Incentivizing Reasoning Capability in LLMs via Reinforcement Learning","version":2},"cited_work":{"arxiv_id":"2501.12948","doi":"10.1016/j.artmed.2024.103001","metadata_source":"pith","pith_arxiv_id":"2501.12948","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"DeepSeek-R1: Incentivizing Reasoning Capability in LLMs via Reinforcement Learning","venue":"cs.CL","work_id":"e6b75ad5-2877-4168-97c8-710407094d20","year":2025},"citing_paper":{"arxiv_id":"2605.19775","last_updated":"2026-05-19T12:43:51Z","snapshot_observed_at":"2026-08-08T23:12:58.922286Z","submitted_at":"2026-05-19T12:43:51Z","title":"Understanding Inference Scaling for LLMs: Bottlenecks, Trade-offs, and Performance Principles","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-05-20T02:11:26.925234Z"},"links":{"cited_paper":"/paper/2501.12948","citing_paper":"/paper/2605.19775"},"observation_digest":"sha256:4dc6414079b3ee3de579bacfd8030014ca3a0366cb9228ce790a1227fb9bd2dd","observation_id":"67452545-d618-444e-95e7-db09289c1fa3","resolution":{"observed_at":"2026-05-20T02:12:58.251061Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Accelerating LLM inference throughput via asynchronous KV cache prefetching","venue":null,"work_id":"e2d707e3-9dc2-4fce-867a-717f2fb6b088","year":2026},"citing_paper":{"arxiv_id":"2605.19775","last_updated":"2026-05-19T12:43:51Z","snapshot_observed_at":"2026-08-08T23:12:58.922286Z","submitted_at":"2026-05-19T12:43:51Z","title":"Understanding Inference Scaling for LLMs: Bottlenecks, Trade-offs, and Performance Principles","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-05-20T02:11:26.925234Z"},"links":{"citing_paper":"/paper/2605.19775"},"observation_digest":"sha256:531107b6cb07235b4d141007c610175eb5397c03bb02a05c394269f394c720f0","observation_id":"1bce6f5a-fda1-45f4-a32e-694187999c6a","resolution":{"observed_at":"2026-05-20T02:12:58.776243Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.21783","last_updated":"2024-11-23T23:27:33Z","snapshot_observed_at":"2026-08-13T17:20:44.002518Z","submitted_at":"2024-07-31T17:54:27Z","title":"The Llama 3 Herd of Models","version":3},"cited_work":{"arxiv_id":"2407.21783","doi":"10.1016/s0749-0720(15","metadata_source":"pith","pith_arxiv_id":"2407.21783","snapshot_observed_at":"2026-07-11T11:50:26.030339Z","title":"The Llama 3 Herd of Models","venue":"cs.AI","work_id":"1549a635-88af-4ac1-acfe-51ae7bb53345","year":2024},"citing_paper":{"arxiv_id":"2605.19775","last_updated":"2026-05-19T12:43:51Z","snapshot_observed_at":"2026-08-08T23:12:58.922286Z","submitted_at":"2026-05-19T12:43:51Z","title":"Understanding Inference Scaling for LLMs: Bottlenecks, Trade-offs, and Performance Principles","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-05-20T02:11:26.925234Z"},"links":{"cited_paper":"/paper/2407.21783","citing_paper":"/paper/2605.19775"},"observation_digest":"sha256:62f7a887437fed98150ba8122b6e005b0c88932bf09b340d4d5b2cad5e55a51c","observation_id":"4d9900ce-7436-4e07-95fb-61b603f96a7b","resolution":{"observed_at":"2026-05-20T02:12:58.256516Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Nvidia dynamo","venue":null,"work_id":"2a470682-bcee-4ea1-96f6-d1c92626524a","year":2025},"citing_paper":{"arxiv_id":"2605.19775","last_updated":"2026-05-19T12:43:51Z","snapshot_observed_at":"2026-08-08T23:12:58.922286Z","submitted_at":"2026-05-19T12:43:51Z","title":"Understanding Inference Scaling for LLMs: Bottlenecks, Trade-offs, and Performance Principles","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-05-20T02:11:26.925234Z"},"links":{"citing_paper":"/paper/2605.19775"},"observation_digest":"sha256:5ed7476800f82262a868c909dfcc3ad7a7aa2dcbe846df7332ec1559d8e7cbc8","observation_id":"8ac2906b-e0cf-4bec-a6ad-817316c66a7f","resolution":{"observed_at":"2026-05-20T02:12:58.747390Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Gpipe: Efficient training of giant neural networks using pipeline parallelism","venue":null,"work_id":"14bbc95b-f855-4e2f-9fb9-f87157d16b69","year":2019},"citing_paper":{"arxiv_id":"2605.19775","last_updated":"2026-05-19T12:43:51Z","snapshot_observed_at":"2026-08-08T23:12:58.922286Z","submitted_at":"2026-05-19T12:43:51Z","title":"Understanding Inference Scaling for LLMs: Bottlenecks, Trade-offs, and Performance Principles","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-05-20T02:11:26.925234Z"},"links":{"citing_paper":"/paper/2605.19775"},"observation_digest":"sha256:466dd5eae8d7b55691adfe4d73ec42ea9b42b0803748c2ea32fb21cdebbbda79","observation_id":"3537f3d1-3d01-4b54-acdf-9fd19ef8cb21","resolution":{"observed_at":"2026-05-20T02:12:58.751430Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Calculon: a methodology and tool for high-level co-design of systems and large language models","venue":null,"work_id":"a43464f6-cf4e-42ae-b750-8ae90a5efe05","year":2023},"citing_paper":{"arxiv_id":"2605.19775","last_updated":"2026-05-19T12:43:51Z","snapshot_observed_at":"2026-08-08T23:12:58.922286Z","submitted_at":"2026-05-19T12:43:51Z","title":"Understanding Inference Scaling for LLMs: Bottlenecks, Trade-offs, and Performance Principles","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-05-20T02:11:26.925234Z"},"links":{"citing_paper":"/paper/2605.19775"},"observation_digest":"sha256:de1cdcdd1d48c410dcf64c37ff57b68551b98995f36943bc305eb85dba48591a","observation_id":"f324fd4c-f6a8-48c0-913d-169da9adea20","resolution":{"observed_at":"2026-05-20T02:12:58.794436Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Efficient memory management for large language model serving with pagedattention","venue":null,"work_id":"592ee0b3-016f-4cad-be1d-280f811079cf","year":2023},"citing_paper":{"arxiv_id":"2605.19775","last_updated":"2026-05-19T12:43:51Z","snapshot_observed_at":"2026-08-08T23:12:58.922286Z","submitted_at":"2026-05-19T12:43:51Z","title":"Understanding Inference Scaling for LLMs: Bottlenecks, Trade-offs, and Performance Principles","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-05-20T02:11:26.925234Z"},"links":{"citing_paper":"/paper/2605.19775"},"observation_digest":"sha256:648159d20c0c3fa72218a72dc68f2a0bc9f37a7bc90eef00f8abc4d7366e29d8","observation_id":"c70d84ad-dc6d-4f3c-b67c-4eed754daada","resolution":{"observed_at":"2026-05-20T02:12:58.759844Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Llm inference serving: Survey of recent advances and opportunities","venue":null,"work_id":"9d624c8d-fa13-4386-95fb-4bad2cd45772","year":2024},"citing_paper":{"arxiv_id":"2605.19775","last_updated":"2026-05-19T12:43:51Z","snapshot_observed_at":"2026-08-08T23:12:58.922286Z","submitted_at":"2026-05-19T12:43:51Z","title":"Understanding Inference Scaling for LLMs: Bottlenecks, Trade-offs, and Performance Principles","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-05-20T02:11:26.925234Z"},"links":{"citing_paper":"/paper/2605.19775"},"observation_digest":"sha256:05198d97f953dd1760efc5195a609c5c01b98f6a5efb4f6ee2e233c912d5d480","observation_id":"5ed0f790-0cce-414c-85b2-f320cdaec115","resolution":{"observed_at":"2026-05-20T02:12:58.798791Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.19442","last_updated":"2025-07-30T05:24:46Z","snapshot_observed_at":"2026-08-15T13:24:51.697670Z","submitted_at":"2024-12-27T04:17:57Z","title":"A Survey on Large Language Model Acceleration based on KV Cache Management","version":3},"cited_work":{"arxiv_id":"2412.19442","doi":"10.48550/arxiv.2412.19442","metadata_source":"pith","pith_arxiv_id":"2412.19442","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"A survey on large lan- guage model acceleration based on kv cache management","venue":"cs.AI","work_id":"5ad86189-256c-4911-bb00-fbfc90ae9a4e","year":2024},"citing_paper":{"arxiv_id":"2605.19775","last_updated":"2026-05-19T12:43:51Z","snapshot_observed_at":"2026-08-08T23:12:58.922286Z","submitted_at":"2026-05-19T12:43:51Z","title":"Understanding Inference Scaling for LLMs: Bottlenecks, Trade-offs, and Performance Principles","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-05-20T02:11:26.925234Z"},"links":{"cited_paper":"/paper/2412.19442","citing_paper":"/paper/2605.19775"},"observation_digest":"sha256:d17b757a9432434df32468c11f6048d6d476c8e97e5c852306fdb56cd3192b6f","observation_id":"32936c06-218e-4973-a7ac-2cc489621af9","resolution":{"observed_at":"2026-05-20T02:12:58.261881Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-07-10T18:20:05.433835+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-10T18:20:05.433835+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Deepseek-v3 technical report","venue":null,"work_id":"7282465a-d764-467d-bc68-7b73769ea0c9","year":2024},"citing_paper":{"arxiv_id":"2605.19775","last_updated":"2026-05-19T12:43:51Z","snapshot_observed_at":"2026-08-08T23:12:58.922286Z","submitted_at":"2026-05-19T12:43:51Z","title":"Understanding Inference Scaling for LLMs: Bottlenecks, Trade-offs, and Performance Principles","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-05-20T02:11:26.925234Z"},"links":{"citing_paper":"/paper/2605.19775"},"observation_digest":"sha256:d11823164fa52f15b34515d81d3f2c795fb8d166aae76f103fa9113d7535ba86","observation_id":"3d06ced6-091a-4814-acc9-9100f6319475","resolution":{"observed_at":"2026-05-20T02:12:58.803051Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Minicache: Kv cache compression in depth dimension for large language models","venue":null,"work_id":"614306e5-089a-425a-b913-136bb0b5a27b","year":2024},"citing_paper":{"arxiv_id":"2605.19775","last_updated":"2026-05-19T12:43:51Z","snapshot_observed_at":"2026-08-08T23:12:58.922286Z","submitted_at":"2026-05-19T12:43:51Z","title":"Understanding Inference Scaling for LLMs: Bottlenecks, Trade-offs, and Performance Principles","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-05-20T02:11:26.925234Z"},"links":{"citing_paper":"/paper/2605.19775"},"observation_digest":"sha256:e8548274c7c6fdc8390704092584b2de24ceb2def473955937c8719861ac2728","observation_id":"23e374e1-46ba-46fc-a19a-a27bd1e04bdc","resolution":{"observed_at":"2026-05-20T02:12:58.823757Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Mlp-offload: Multi-level, multi-path offloading for llm pre-training to break the gpu memory wall","venue":null,"work_id":"63c58b9b-b826-480a-8013-4f54f24f0c69","year":2025},"citing_paper":{"arxiv_id":"2605.19775","last_updated":"2026-05-19T12:43:51Z","snapshot_observed_at":"2026-08-08T23:12:58.922286Z","submitted_at":"2026-05-19T12:43:51Z","title":"Understanding Inference Scaling for LLMs: Bottlenecks, Trade-offs, and Performance Principles","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-05-20T02:11:26.925234Z"},"links":{"citing_paper":"/paper/2605.19775"},"observation_digest":"sha256:9ff3068ff163651890f1e01ca01e24a38dfb570eece29c0476d52f14f440ed4a","observation_id":"e2754aa6-4304-4c72-8865-4a43624110c5","resolution":{"observed_at":"2026-05-20T02:12:58.811075Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Openai o1 system card","venue":null,"work_id":"1f53e8d0-3b93-47f2-9661-bfaf3e7f9126","year":2024},"citing_paper":{"arxiv_id":"2605.19775","last_updated":"2026-05-19T12:43:51Z","snapshot_observed_at":"2026-08-08T23:12:58.922286Z","submitted_at":"2026-05-19T12:43:51Z","title":"Understanding Inference Scaling for LLMs: Bottlenecks, Trade-offs, and Performance Principles","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-05-20T02:11:26.925234Z"},"links":{"citing_paper":"/paper/2605.19775"},"observation_digest":"sha256:ddfb5b558a16850929ed55295411b65fe1224aaaa384a2575b870a752139a9ee","observation_id":"eec82d19-bac2-4f3a-90a9-3fc0ccc63dce","resolution":{"observed_at":"2026-05-20T02:12:58.807002Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2505.01658","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"A survey on inference engines for large language models: Perspectives on optimization and efficiency","venue":null,"work_id":"7d6da7dc-3212-4a0b-80b2-17fd68881bc5","year":2025},"citing_paper":{"arxiv_id":"2605.19775","last_updated":"2026-05-19T12:43:51Z","snapshot_observed_at":"2026-08-08T23:12:58.922286Z","submitted_at":"2026-05-19T12:43:51Z","title":"Understanding Inference Scaling for LLMs: Bottlenecks, Trade-offs, and Performance Principles","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-05-20T02:11:26.925234Z"},"links":{"citing_paper":"/paper/2605.19775"},"observation_digest":"sha256:b76e2bf07fc1bc2971869b6d84cd0a2b78d02af5953c03e46caaba846e6fc690","observation_id":"7c14ce41-d42b-4b04-b685-b6e78eb71740","resolution":{"observed_at":"2026-05-20T02:12:58.267265Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Mooncake: A kvcache-centric disaggregated architecture for llm serving","venue":null,"work_id":"ba6f202c-20f6-4a10-8842-634ab4aada26","year":2024},"citing_paper":{"arxiv_id":"2605.19775","last_updated":"2026-05-19T12:43:51Z","snapshot_observed_at":"2026-08-08T23:12:58.922286Z","submitted_at":"2026-05-19T12:43:51Z","title":"Understanding Inference Scaling for LLMs: Bottlenecks, Trade-offs, and Performance Principles","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-05-20T02:11:26.925234Z"},"links":{"citing_paper":"/paper/2605.19775"},"observation_digest":"sha256:37ecd4ee4c1888fc69832d3dcaa815c0a2c24951b568b1d069c8dbde93f8d0da","observation_id":"46755b82-94d7-4a76-b492-25849a04a9f3","resolution":{"observed_at":"2026-05-20T02:12:58.785903Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Prophet: An llm infer- ence engine optimized for head-of-line blocking","venue":null,"work_id":"6366b1da-76e0-40ed-96ac-98c77aaafe73","year":2024},"citing_paper":{"arxiv_id":"2605.19775","last_updated":"2026-05-19T12:43:51Z","snapshot_observed_at":"2026-08-08T23:12:58.922286Z","submitted_at":"2026-05-19T12:43:51Z","title":"Understanding Inference Scaling for LLMs: Bottlenecks, Trade-offs, and Performance Principles","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-05-20T02:11:26.925234Z"},"links":{"citing_paper":"/paper/2605.19775"},"observation_digest":"sha256:ab39f7eef794c9e754b54868ffdeaee6238ac0f5d3cacbe9de0112f67d6d06e9","observation_id":"6843e24d-0796-440a-9bb1-dd6ec58d8003","resolution":{"observed_at":"2026-05-20T02:12:58.819523Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Hbf: High bandwidth flash","venue":null,"work_id":"a1f5a139-2a12-4153-bd2d-321e4337e5ea","year":2025},"citing_paper":{"arxiv_id":"2605.19775","last_updated":"2026-05-19T12:43:51Z","snapshot_observed_at":"2026-08-08T23:12:58.922286Z","submitted_at":"2026-05-19T12:43:51Z","title":"Understanding Inference Scaling for LLMs: Bottlenecks, Trade-offs, and Performance Principles","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-05-20T02:11:26.925234Z"},"links":{"citing_paper":"/paper/2605.19775"},"observation_digest":"sha256:5b8fdae3c7e3e9efbd27845f6478a50824ed4d9efa1bcd10ca3e61fc51495c8b","observation_id":"09a3c6a1-a70c-4365-8c11-c719c8b0b962","resolution":{"observed_at":"2026-05-20T02:12:58.815200Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"1909.08053","last_updated":"2020-03-13T23:45:18Z","snapshot_observed_at":"2026-08-12T10:50:46.357243Z","submitted_at":"2019-09-17T19:42:54Z","title":"Megatron-LM: Training Multi-Billion Parameter Language Models Using Model Parallelism","version":4},"cited_work":{"arxiv_id":"1909.08053","doi":"10.48550/arxiv.1909.08053","metadata_source":"pith","pith_arxiv_id":"1909.08053","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Megatron-LM: Training Multi-Billion Parameter Language Models Using Model Parallelism","venue":"cs.CL","work_id":"c888e6d1-0b1d-43d6-9ef5-f0912a0efa1b","year":2019},"citing_paper":{"arxiv_id":"2605.19775","last_updated":"2026-05-19T12:43:51Z","snapshot_observed_at":"2026-08-08T23:12:58.922286Z","submitted_at":"2026-05-19T12:43:51Z","title":"Understanding Inference Scaling for LLMs: Bottlenecks, Trade-offs, and Performance Principles","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-05-20T02:11:26.925234Z"},"links":{"cited_paper":"/paper/1909.08053","citing_paper":"/paper/2605.19775"},"observation_digest":"sha256:eaef8dc8b77ed920c73e901dce2f9c618e271e7ef2a87562f7ff82052b5dd3f1","observation_id":"391ab928-c06c-4c0c-ad09-c8546f47df1d","resolution":{"observed_at":"2026-05-20T02:12:58.276412Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-07-09T10:48:33.392193+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-09T10:48:33.392193+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Mechanistic interpretability of attention heads in reasoning llms","venue":null,"work_id":"1595976e-8ee3-4463-bf65-70d8836ca6d5","year":2025},"citing_paper":{"arxiv_id":"2605.19775","last_updated":"2026-05-19T12:43:51Z","snapshot_observed_at":"2026-08-08T23:12:58.922286Z","submitted_at":"2026-05-19T12:43:51Z","title":"Understanding Inference Scaling for LLMs: Bottlenecks, Trade-offs, and Performance Principles","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-05-20T02:11:26.925234Z"},"links":{"citing_paper":"/paper/2605.19775"},"observation_digest":"sha256:ae19664162d104d8c493dcacd41dff59bfc8cf6a29af6f926581fc23852401ee","observation_id":"ec55eb53-8863-414b-b09b-715fa2a859b9","resolution":{"observed_at":"2026-05-20T02:12:58.764292Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Chain-of-thought prompting elicits reasoning in large language mod- els","venue":null,"work_id":"6f7ccc01-f5d2-4514-9902-2f550547137c","year":2022},"citing_paper":{"arxiv_id":"2605.19775","last_updated":"2026-05-19T12:43:51Z","snapshot_observed_at":"2026-08-08T23:12:58.922286Z","submitted_at":"2026-05-19T12:43:51Z","title":"Understanding Inference Scaling for LLMs: Bottlenecks, Trade-offs, and Performance Principles","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-05-20T02:11:26.925234Z"},"links":{"citing_paper":"/paper/2605.19775"},"observation_digest":"sha256:1362770a919394fa495935ac73508d9e499617361456c911813a77db5275bb63","observation_id":"af623f19-d7cf-47db-b58b-dab28d57fc1c","resolution":{"observed_at":"2026-05-20T02:12:58.733884Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.09686","last_updated":"2025-01-23T08:44:44Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-01-16T17:37:58Z","title":"Towards Large Reasoning Models: A Survey of Reinforced Reasoning with Large Language Models","version":3},"cited_work":{"arxiv_id":"2501.09686","doi":"10.48550/arxiv.2501.09686","metadata_source":"pith","pith_arxiv_id":"2501.09686","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Towards Large Reasoning Models: A Survey of Reinforced Reasoning with Large Language Models","venue":"cs.AI","work_id":"9b0d5273-2b7c-477e-b49b-2e0edab55456","year":2025},"citing_paper":{"arxiv_id":"2605.19775","last_updated":"2026-05-19T12:43:51Z","snapshot_observed_at":"2026-08-08T23:12:58.922286Z","submitted_at":"2026-05-19T12:43:51Z","title":"Understanding Inference Scaling for LLMs: Bottlenecks, Trade-offs, and Performance Principles","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-05-20T02:11:26.925234Z"},"links":{"cited_paper":"/paper/2501.09686","citing_paper":"/paper/2605.19775"},"observation_digest":"sha256:671027ba91f86c3ff73b8b989a880a30655d86f4df96af73b02044222a419cd3","observation_id":"1e349555-e289-4cd3-9c23-86a6dac0275f","resolution":{"observed_at":"2026-05-20T02:12:58.287921Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Characterizing the behavior and impact of kv caching on transformer inferences under concurrency","venue":null,"work_id":"fe6643ef-729d-46f0-9d18-b2c459c02539","year":2025},"citing_paper":{"arxiv_id":"2605.19775","last_updated":"2026-05-19T12:43:51Z","snapshot_observed_at":"2026-08-08T23:12:58.922286Z","submitted_at":"2026-05-19T12:43:51Z","title":"Understanding Inference Scaling for LLMs: Bottlenecks, Trade-offs, and Performance Principles","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-05-20T02:11:26.925234Z"},"links":{"citing_paper":"/paper/2605.19775"},"observation_digest":"sha256:246b11bed4a6c1f8e16fe1ea8714ba27b4efd7a22e5ee02155ae1083d8497aa7","observation_id":"d943e6d4-c4c7-49da-b37d-7fac4718f538","resolution":{"observed_at":"2026-05-20T02:12:58.739269Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2502.13124","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-04T20:40:08.275804Z","title":"Naturalreasoning: Reasoning in the wild with 2.8 m challenging questions","venue":null,"work_id":"d185ff64-8137-4c75-a12b-852746b6bd58","year":2025},"citing_paper":{"arxiv_id":"2605.19775","last_updated":"2026-05-19T12:43:51Z","snapshot_observed_at":"2026-08-08T23:12:58.922286Z","submitted_at":"2026-05-19T12:43:51Z","title":"Understanding Inference Scaling for LLMs: Bottlenecks, Trade-offs, and Performance Principles","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-05-20T02:11:26.925234Z"},"links":{"citing_paper":"/paper/2605.19775"},"observation_digest":"sha256:03e942c740d5fafe60de131b9491b2d36b001f9adfa39e5281c7a949de54d850","observation_id":"4837cd02-1a1f-4be7-a555-c9a6b6b3e2ce","resolution":{"observed_at":"2026-05-20T02:12:58.283403Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Insights into deepseek-v3: Scaling challenges and reflections on hardware for ai architectures","venue":null,"work_id":"d7f8916d-9af8-4982-92d4-1c0d37052b36","year":2025},"citing_paper":{"arxiv_id":"2605.19775","last_updated":"2026-05-19T12:43:51Z","snapshot_observed_at":"2026-08-08T23:12:58.922286Z","submitted_at":"2026-05-19T12:43:51Z","title":"Understanding Inference Scaling for LLMs: Bottlenecks, Trade-offs, and Performance Principles","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-05-20T02:11:26.925234Z"},"links":{"citing_paper":"/paper/2605.19775"},"observation_digest":"sha256:5d1882d4e749cb6155412f555b04cbe88914f4818529666cfb6459fc67378dba","observation_id":"2058e99c-1c98-459c-ab03-76f14c1e7216","resolution":{"observed_at":"2026-05-20T02:12:58.729560Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-18T06:34:40.430872+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2605.19775","last_updated":"2026-05-19T12:43:51Z","latest_version":1,"primary_category":"cs.DC","snapshot_observed_at":"2026-08-08T23:12:58.922286Z","submitted_at":"2026-05-19T12:43:51Z","title":"Understanding Inference Scaling for LLMs: Bottlenecks, Trade-offs, and Performance Principles"},"reference_resolution":{"displayed":40,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":13,"verified_fuzzy":27},"total_outbound_references":40},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-18T06:34:40.430872+00:00","source":"crossref"},{"observed_at":"2026-08-18T06:34:34.496301+00:00","source":"retraction_watch"}],"thesis":"As of 18 August 2026, this Paper Citation Record lists 40 of 40 outbound references and 2 inbound Pith citation observations for arXiv:2605.19775."}