{"as_of":"2026-08-08T02:15:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:69a6440ac56a37281fe20e3757b6318188b02d677d08f9ad383c6aa164912dbd","coverage":[{"denominator":36,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":36,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T14:16:47.986431Z","state":"measured"},{"denominator":37,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":37,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-07T06:34:17.273281+00:00","state":"measured"},{"denominator":1,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":1,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-02T08:31:01.469310Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2505.19481","last_updated":"2025-05-26T04:03:48Z","snapshot_observed_at":"2026-08-07T14:11:12.334794Z","submitted_at":"2025-05-26T04:03:48Z","title":"Win Fast or Lose Slow: Balancing Speed and Accuracy in Latency-Sensitive Decisions of LLMs","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.19481","snapshot_observed_at":"2026-08-02T08:31:01.469310Z","title":"Win fast or lose slow: Balancing speed and accuracy in latency-sensitive decisions of LLMs.arXiv preprint arXiv:2505.19481,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.05690","last_updated":"2026-07-19T20:21:45Z","snapshot_observed_at":"2026-08-02T15:10:53.788869Z","submitted_at":"2026-07-06T23:16:11Z","title":"Memory in the Loop: In-Process Retrieval as Extended Working Memory for Language Agents","version":2},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-02T08:31:01.469310Z"},"links":{"cited_paper":"/paper/2505.19481","citing_paper":"/paper/2607.05690"},"observation_digest":"sha256:43ef385007bfca196aca43f8e46bce9a2c9e29ed36bf824a13121a5df7e07ad0","observation_id":"5fd66efb-c8a1-42e6-9e11-6413dbdd1f35","resolution":{"observed_at":"2026-08-02T08:31:01.469310Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2505.19481/citation-record","integrity":"/paper/2505.19481/integrity","json":"/paper/2505.19481/citation-record.json","paper":"/paper/2505.19481"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2412.08905","last_updated":"2024-12-12T03:37:41Z","snapshot_observed_at":"2026-08-05T04:04:21.846023Z","submitted_at":"2024-12-12T03:37:41Z","title":"Phi-4 Technical Report","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.08905","snapshot_observed_at":"2026-08-07T14:16:44.685301Z","title":"Phi-4 technical report","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.19481","last_updated":"2025-05-26T04:03:48Z","snapshot_observed_at":"2026-08-07T14:11:12.334794Z","submitted_at":"2025-05-26T04:03:48Z","title":"Win Fast or Lose Slow: Balancing Speed and Accuracy in Latency-Sensitive Decisions of LLMs","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-07T14:16:44.685301Z"},"links":{"cited_paper":"/paper/2412.08905","citing_paper":"/paper/2505.19481"},"observation_digest":"sha256:fbb290a2bc66f633cf7921d3086943d49f7f65a60a179dcbbf565f9ceccc3a9c","observation_id":"84d2d838-09e4-437a-bb71-a379e9e77c1c","resolution":{"observed_at":"2026-08-07T14:16:44.685301Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:16:50.206127Z","title":"Risk and return in high- frequency trading.Journal of Financial and Quantitative Analysis,54993–1024","venue":null,"work_id":"f044e6bb-8ec9-4252-bedf-f6c73e8284c4","year":2019},"citing_paper":{"arxiv_id":"2505.19481","last_updated":"2025-05-26T04:03:48Z","snapshot_observed_at":"2026-08-07T14:11:12.334794Z","submitted_at":"2025-05-26T04:03:48Z","title":"Win Fast or Lose Slow: Balancing Speed and Accuracy in Latency-Sensitive Decisions of LLMs","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-07T14:16:44.754187Z"},"links":{"citing_paper":"/paper/2505.19481"},"observation_digest":"sha256:cc92caae93b137761fe8ad0c2787fb52a0778fce5fa3dbd85fe388080c9d4e69","observation_id":"3ddd0015-a0aa-45d3-930b-5045c122ada8","resolution":{"observed_at":"2026-08-07T14:16:50.282112Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.14325","last_updated":"2023-05-23T17:55:11Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-23T17:55:11Z","title":"Improving Factuality and Reasoning in Language Models through Multiagent Debate","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.14325","snapshot_observed_at":"2026-08-07T14:16:44.824215Z","title":"B.andMordatch, I.(2023)","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.19481","last_updated":"2025-05-26T04:03:48Z","snapshot_observed_at":"2026-08-07T14:11:12.334794Z","submitted_at":"2025-05-26T04:03:48Z","title":"Win Fast or Lose Slow: Balancing Speed and Accuracy in Latency-Sensitive Decisions of LLMs","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-07T14:16:44.824215Z"},"links":{"cited_paper":"/paper/2305.14325","citing_paper":"/paper/2505.19481"},"observation_digest":"sha256:12f4b57384fec4eb5efdfa073fe20716d415437a793c7ab8ee9c4f719efb4c60","observation_id":"2c98747c-c0a5-448b-aa32-ccfaadd69797","resolution":{"observed_at":"2026-08-07T14:16:44.824215Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:16:50.012865Z","title":"E.(1967)","venue":null,"work_id":"b5200995-1be7-4f79-9315-c83cf27495c7","year":1967},"citing_paper":{"arxiv_id":"2505.19481","last_updated":"2025-05-26T04:03:48Z","snapshot_observed_at":"2026-08-07T14:11:12.334794Z","submitted_at":"2025-05-26T04:03:48Z","title":"Win Fast or Lose Slow: Balancing Speed and Accuracy in Latency-Sensitive Decisions of LLMs","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-07T14:16:44.909069Z"},"links":{"citing_paper":"/paper/2505.19481"},"observation_digest":"sha256:0c0b9c87e50318d50276f9c460b4b4d7902f8c7f667d6d74521cb041d01d4e58","observation_id":"a51b0919-7104-4e7b-a15e-6b35930f6708","resolution":{"observed_at":"2026-08-07T14:16:50.123459Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2210.17323","last_updated":"2023-03-22T13:10:47Z","snapshot_observed_at":"2026-08-07T08:38:54.025062Z","submitted_at":"2022-10-31T13:42:40Z","title":"GPTQ: Accurate Post-Training Quantization for Generative Pre-trained Transformers","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2210.17323","snapshot_observed_at":"2026-08-07T14:16:45.026717Z","title":"Gptq: Accurate post-training quantization for generative pre-trained transformers","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.19481","last_updated":"2025-05-26T04:03:48Z","snapshot_observed_at":"2026-08-07T14:11:12.334794Z","submitted_at":"2025-05-26T04:03:48Z","title":"Win Fast or Lose Slow: Balancing Speed and Accuracy in Latency-Sensitive Decisions of LLMs","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-07T14:16:45.026717Z"},"links":{"cited_paper":"/paper/2210.17323","citing_paper":"/paper/2505.19481"},"observation_digest":"sha256:a00aa931a12528d812535c058f63f963871c35b7197f93766f279f98b1076495","observation_id":"2109fd91-028e-47fc-acf6-72e90429ab73","resolution":{"observed_at":"2026-08-07T14:16:45.026717Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:16:49.848733Z","title":"Reinforcement Learning Equilibrium in Limit Order Markets.Journal of Economic Dynamics and Control,144","venue":null,"work_id":"01978a26-685b-4c2b-9381-8919ad365ff0","year":2022},"citing_paper":{"arxiv_id":"2505.19481","last_updated":"2025-05-26T04:03:48Z","snapshot_observed_at":"2026-08-07T14:11:12.334794Z","submitted_at":"2025-05-26T04:03:48Z","title":"Win Fast or Lose Slow: Balancing Speed and Accuracy in Latency-Sensitive Decisions of LLMs","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-07T14:16:45.172356Z"},"links":{"citing_paper":"/paper/2505.19481"},"observation_digest":"sha256:5cc78c7e5b259753b40796605550ddb846439472f292ee145a0661ff187624fd","observation_id":"0a04d155-2ff9-4621-967e-c0eac87d9acf","resolution":{"observed_at":"2026-08-07T14:16:49.925155Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.18079","last_updated":"2025-05-28T18:58:29Z","snapshot_observed_at":"2026-08-05T09:48:25.127042Z","submitted_at":"2024-01-31T18:58:14Z","title":"KVQuant: Towards 10 Million Context Length LLM Inference with KV Cache Quantization","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.18079","snapshot_observed_at":"2026-08-07T14:16:45.310548Z","title":"W .,Shao, Y","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.19481","last_updated":"2025-05-26T04:03:48Z","snapshot_observed_at":"2026-08-07T14:11:12.334794Z","submitted_at":"2025-05-26T04:03:48Z","title":"Win Fast or Lose Slow: Balancing Speed and Accuracy in Latency-Sensitive Decisions of LLMs","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-07T14:16:45.310548Z"},"links":{"cited_paper":"/paper/2401.18079","citing_paper":"/paper/2505.19481"},"observation_digest":"sha256:e36c004c195abee267ce4069b49f66b1e56d35c749d7bce8f9653551d04d5200","observation_id":"061c7b6a-8acf-416f-8cd8-aeb6b27e50cf","resolution":{"observed_at":"2026-08-07T14:16:45.310548Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.08585","last_updated":"2024-12-17T05:40:09Z","snapshot_observed_at":"2026-07-06T20:05:23.685854Z","submitted_at":"2024-12-11T18:03:05Z","title":"TurboAttention: Efficient Attention Approximation For High Throughputs LLMs","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.08585","snapshot_observed_at":"2026-08-07T14:16:45.426202Z","title":"Turboat- tention: Efficient attention approximation for high throughputs llms","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.19481","last_updated":"2025-05-26T04:03:48Z","snapshot_observed_at":"2026-08-07T14:11:12.334794Z","submitted_at":"2025-05-26T04:03:48Z","title":"Win Fast or Lose Slow: Balancing Speed and Accuracy in Latency-Sensitive Decisions of LLMs","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-07T14:16:45.426202Z"},"links":{"cited_paper":"/paper/2412.08585","citing_paper":"/paper/2505.19481"},"observation_digest":"sha256:8cead5d2fa640b50b41c1bb49743c4c02d302a8deb62b7aa8894bca8a93fcefb","observation_id":"5ef6b394-a704-4f99-9dea-3ab8253335be","resolution":{"observed_at":"2026-08-07T14:16:45.426202Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.05527","last_updated":"2024-09-30T22:44:58Z","snapshot_observed_at":"2026-08-07T12:20:18.370756Z","submitted_at":"2024-03-08T18:48:30Z","title":"GEAR: An Efficient KV Cache Compression Recipe for Near-Lossless Generative Inference of LLM","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.05527","snapshot_observed_at":"2026-08-07T14:16:45.550536Z","title":"Gear: An efficient kv cache compression recipe for near-lossless generative inference of llm","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.19481","last_updated":"2025-05-26T04:03:48Z","snapshot_observed_at":"2026-08-07T14:11:12.334794Z","submitted_at":"2025-05-26T04:03:48Z","title":"Win Fast or Lose Slow: Balancing Speed and Accuracy in Latency-Sensitive Decisions of LLMs","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-07T14:16:45.550536Z"},"links":{"cited_paper":"/paper/2403.05527","citing_paper":"/paper/2505.19481"},"observation_digest":"sha256:6e6da242e8ea5eed8e3dcaa86ae1f301551ea0e5ce78da87d71c37fd5e8807a6","observation_id":"704c6af7-e828-4af3-a14c-e0b581fc656f","resolution":{"observed_at":"2026-08-07T14:16:45.550536Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:16:49.671472Z","title":"M.,Uszkoreit, J.,Le, Q.andPetrov, S.(2019)","venue":null,"work_id":"6cd01930-1270-4bff-9ca9-ac048aa303be","year":2019},"citing_paper":{"arxiv_id":"2505.19481","last_updated":"2025-05-26T04:03:48Z","snapshot_observed_at":"2026-08-07T14:11:12.334794Z","submitted_at":"2025-05-26T04:03:48Z","title":"Win Fast or Lose Slow: Balancing Speed and Accuracy in Latency-Sensitive Decisions of LLMs","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-07T14:16:45.633428Z"},"links":{"citing_paper":"/paper/2505.19481"},"observation_digest":"sha256:683ecd6bef582ffe0fb9babf95714ff0d1f04f5e4f02f7bfbb68dddd1ee529f6","observation_id":"7784396c-b9d5-4298-b30f-27e412081587","resolution":{"observed_at":"2026-08-07T14:16:49.780697Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2309.06180","last_updated":"2023-09-12T12:50:04Z","snapshot_observed_at":"2026-08-02T09:51:08.145755Z","submitted_at":"2023-09-12T12:50:04Z","title":"Efficient Memory Management for Large Language Model Serving with PagedAttention","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.06180","snapshot_observed_at":"2026-08-07T14:16:45.756234Z","title":"H.,Gonzalez, J","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.19481","last_updated":"2025-05-26T04:03:48Z","snapshot_observed_at":"2026-08-07T14:11:12.334794Z","submitted_at":"2025-05-26T04:03:48Z","title":"Win Fast or Lose Slow: Balancing Speed and Accuracy in Latency-Sensitive Decisions of LLMs","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-07T14:16:45.756234Z"},"links":{"cited_paper":"/paper/2309.06180","citing_paper":"/paper/2505.19481"},"observation_digest":"sha256:8666a5a0bab1aad9df5b6b5fb0fb0844f7db4e238082f29fd77ce4b9333a795b","observation_id":"db4c5c80-bcd4-4458-8166-3f2475342433","resolution":{"observed_at":"2026-08-07T14:16:45.756234Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2306.02272","last_updated":"2024-01-24T02:53:27Z","snapshot_observed_at":"2026-07-06T15:37:39.433980Z","submitted_at":"2023-06-04T06:33:13Z","title":"OWQ: Outlier-Aware Weight Quantization for Efficient Fine-Tuning and Inference of Large Language Models","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.02272","snapshot_observed_at":"2026-08-07T14:16:45.817948Z","title":"Owq: Outlier-aware weight quantization for efficient fine-tuning and inference of large language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.19481","last_updated":"2025-05-26T04:03:48Z","snapshot_observed_at":"2026-08-07T14:11:12.334794Z","submitted_at":"2025-05-26T04:03:48Z","title":"Win Fast or Lose Slow: Balancing Speed and Accuracy in Latency-Sensitive Decisions of LLMs","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-07T14:16:45.817948Z"},"links":{"cited_paper":"/paper/2306.02272","citing_paper":"/paper/2505.19481"},"observation_digest":"sha256:ded11bc0c3083dec2baf3e420370b15ec8d87fd7feebc7952e2faaf13fc591cb","observation_id":"8fb75c14-1379-48bc-bb2b-e1624725b7a9","resolution":{"observed_at":"2026-08-07T14:16:45.817948Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:16:49.516471Z","title":"Camel: Communicative agents for\" mind\" exploration of large language model society.Advances in Neural Information Processing Systems,3651991–52008","venue":null,"work_id":"ac47a161-4026-4a7a-9312-8514a198ebba","year":2023},"citing_paper":{"arxiv_id":"2505.19481","last_updated":"2025-05-26T04:03:48Z","snapshot_observed_at":"2026-08-07T14:11:12.334794Z","submitted_at":"2025-05-26T04:03:48Z","title":"Win Fast or Lose Slow: Balancing Speed and Accuracy in Latency-Sensitive Decisions of LLMs","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-07T14:16:45.906545Z"},"links":{"citing_paper":"/paper/2505.19481"},"observation_digest":"sha256:550c705852b4a1aa018c377bb433ba906e7eb0ca19c538ae3c9f48577f889cb9","observation_id":"bf0183ab-eb04-4312-9e9d-257fab5dba5a","resolution":{"observed_at":"2026-08-07T14:16:49.582197Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:16:46.018741Z","title":"Svdquant: Absorbing outliers by low-rank components for 4-bit diffusion models","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.19481","last_updated":"2025-05-26T04:03:48Z","snapshot_observed_at":"2026-08-07T14:11:12.334794Z","submitted_at":"2025-05-26T04:03:48Z","title":"Win Fast or Lose Slow: Balancing Speed and Accuracy in Latency-Sensitive Decisions of LLMs","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-07T14:16:46.018741Z"},"links":{"citing_paper":"/paper/2505.19481"},"observation_digest":"sha256:684319316bf0258f11ea6bf9c63fd968135bb3c916acc5f0ae8d0e61da132566","observation_id":"cdabc844-48c4-4402-99c3-b3fd05469cb3","resolution":{"observed_at":"2026-08-07T14:16:46.018741Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.04532","last_updated":"2025-05-01T02:14:05Z","snapshot_observed_at":"2026-07-06T18:11:09.767624Z","submitted_at":"2024-05-07T17:59:30Z","title":"QServe: W4A8KV4 Quantization and System Co-design for Efficient LLM Serving","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.04532","snapshot_observed_at":"2026-08-07T14:16:46.099993Z","title":"Qserve: W4a8kv4 quantization and system co-design for efficient llm serving","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.19481","last_updated":"2025-05-26T04:03:48Z","snapshot_observed_at":"2026-08-07T14:11:12.334794Z","submitted_at":"2025-05-26T04:03:48Z","title":"Win Fast or Lose Slow: Balancing Speed and Accuracy in Latency-Sensitive Decisions of LLMs","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-07T14:16:46.099993Z"},"links":{"cited_paper":"/paper/2405.04532","citing_paper":"/paper/2505.19481"},"observation_digest":"sha256:3e72f2976d01c54ba10f44b9d07c766b71adbf1f389bb7071eec79fcad74126a","observation_id":"770390de-bbde-42c8-987f-66949810bf77","resolution":{"observed_at":"2026-08-07T14:16:46.099993Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.11865","last_updated":"2024-06-18T03:07:37Z","snapshot_observed_at":"2026-07-06T17:05:06.923713Z","submitted_at":"2023-12-19T05:27:16Z","title":"Large Language Models Play StarCraft II: Benchmarks and A Chain of Summarization Approach","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.11865","snapshot_observed_at":"2026-08-07T14:16:46.176908Z","title":"Large language models play starcraft ii: Benchmarks and a chain of summarization approach","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.19481","last_updated":"2025-05-26T04:03:48Z","snapshot_observed_at":"2026-08-07T14:11:12.334794Z","submitted_at":"2025-05-26T04:03:48Z","title":"Win Fast or Lose Slow: Balancing Speed and Accuracy in Latency-Sensitive Decisions of LLMs","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-07T14:16:46.176908Z"},"links":{"cited_paper":"/paper/2312.11865","citing_paper":"/paper/2505.19481"},"observation_digest":"sha256:4a47306677113eda14d66c0936ea18de8c7edf5eb2678573c960949b4180cde9","observation_id":"dba7ae56-c724-417d-b318-3223159b3d4a","resolution":{"observed_at":"2026-08-07T14:16:46.176908Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:16:49.332752Z","title":"Pointer sentinel mixture models","venue":null,"work_id":"21a10af9-e885-4a7f-8dbb-c1a40d0cb42a","year":2016},"citing_paper":{"arxiv_id":"2505.19481","last_updated":"2025-05-26T04:03:48Z","snapshot_observed_at":"2026-08-07T14:11:12.334794Z","submitted_at":"2025-05-26T04:03:48Z","title":"Win Fast or Lose Slow: Balancing Speed and Accuracy in Latency-Sensitive Decisions of LLMs","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-07T14:16:46.257128Z"},"links":{"citing_paper":"/paper/2505.19481"},"observation_digest":"sha256:66cf102d64bde5b75532e0302adc10de739e01a8eeff05333d1df750af075e74","observation_id":"91f7cd50-d972-48c5-a496-4a51452519dc","resolution":{"observed_at":"2026-08-07T14:16:49.407735Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2209.05433","last_updated":"2022-09-29T20:47:07Z","snapshot_observed_at":"2026-07-06T13:51:20.183063Z","submitted_at":"2022-09-12T17:39:55Z","title":"FP8 Formats for Deep Learning","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2209.05433","snapshot_observed_at":"2026-08-07T14:16:46.324314Z","title":"Fp8 formats for deep learning","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2505.19481","last_updated":"2025-05-26T04:03:48Z","snapshot_observed_at":"2026-08-07T14:11:12.334794Z","submitted_at":"2025-05-26T04:03:48Z","title":"Win Fast or Lose Slow: Balancing Speed and Accuracy in Latency-Sensitive Decisions of LLMs","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-07T14:16:46.324314Z"},"links":{"cited_paper":"/paper/2209.05433","citing_paper":"/paper/2505.19481"},"observation_digest":"sha256:2852cfe002490f9a36042046d15d169ddcef9d92ad81426ea8a59f0960ad33b1","observation_id":"b98d44a7-93d2-4a11-9e1c-b97f5cae9958","resolution":{"observed_at":"2026-08-07T14:16:46.324314Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2210.10595","last_updated":"2022-10-19T14:39:10Z","snapshot_observed_at":"2026-07-06T14:07:42.787294Z","submitted_at":"2022-10-19T14:39:10Z","title":"DIAMBRA Arena: a New Reinforcement Learning Platform for Research and Experimentation","version":1},"cited_work":{"arxiv_id":"2210.10595","doi":null,"metadata_source":"pith","pith_arxiv_id":"2210.10595","snapshot_observed_at":"2026-08-07T14:16:48.533653Z","title":"DIAMBRA Arena: a New Reinforcement Learning Platform for Research and Experimentation","venue":"cs.LG","work_id":"85559006-7a43-48cc-8edd-7d86a7d1afe9","year":2022},"citing_paper":{"arxiv_id":"2505.19481","last_updated":"2025-05-26T04:03:48Z","snapshot_observed_at":"2026-08-07T14:11:12.334794Z","submitted_at":"2025-05-26T04:03:48Z","title":"Win Fast or Lose Slow: Balancing Speed and Accuracy in Latency-Sensitive Decisions of LLMs","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-07T14:16:46.386384Z"},"links":{"cited_paper":"/paper/2210.10595","citing_paper":"/paper/2505.19481"},"observation_digest":"sha256:4cacdf4f1b440618222137c299a182f6f546b24a3475719eb4a6f4bf5585bd67","observation_id":"8c873734-c063-4d01-b4e4-58e9d511d51c","resolution":{"observed_at":"2026-08-07T14:16:48.580691Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2302.05397","last_updated":"2023-02-10T17:47:54Z","snapshot_observed_at":"2026-08-02T22:31:22.531484Z","submitted_at":"2023-02-10T17:47:54Z","title":"A Practical Mixed Precision Algorithm for Post-Training Quantization","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2302.05397","snapshot_observed_at":"2026-08-07T14:16:46.449162Z","title":"P .,Nagel, M.,van Baalen, M.,Huang, Y .,Patel, C.andBlankevoort, T .(2023)","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.19481","last_updated":"2025-05-26T04:03:48Z","snapshot_observed_at":"2026-08-07T14:11:12.334794Z","submitted_at":"2025-05-26T04:03:48Z","title":"Win Fast or Lose Slow: Balancing Speed and Accuracy in Latency-Sensitive Decisions of LLMs","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-07T14:16:46.449162Z"},"links":{"cited_paper":"/paper/2302.05397","citing_paper":"/paper/2505.19481"},"observation_digest":"sha256:188e4dbcdfe2e02258fcf7f788d6beb955904601aa1f4b90888c4154cfd5807d","observation_id":"c3820474-25b1-4a27-a874-f80bbc3cef51","resolution":{"observed_at":"2026-08-07T14:16:46.449162Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.15115","last_updated":"2025-01-03T02:18:21Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-12-19T17:56:09Z","title":"Qwen2.5 Technical Report","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.15115","snapshot_observed_at":"2026-08-07T14:16:46.527711Z","title":"Qwen2.5 technical report","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.19481","last_updated":"2025-05-26T04:03:48Z","snapshot_observed_at":"2026-08-07T14:11:12.334794Z","submitted_at":"2025-05-26T04:03:48Z","title":"Win Fast or Lose Slow: Balancing Speed and Accuracy in Latency-Sensitive Decisions of LLMs","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-07T14:16:46.527711Z"},"links":{"cited_paper":"/paper/2412.15115","citing_paper":"/paper/2505.19481"},"observation_digest":"sha256:cbd4721bb63ca5ebd1f15cfa0db06ed87c241e3f416ae1574937eb3b6d685b35","observation_id":"d09da9e9-10f6-40ae-9728-7f9ce9385803","resolution":{"observed_at":"2026-08-07T14:16:46.527711Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1902.04043","last_updated":"2019-12-09T07:26:52Z","snapshot_observed_at":"2026-07-06T07:32:24.974529Z","submitted_at":"2019-02-11T18:43:53Z","title":"The StarCraft Multi-Agent Challenge","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1902.04043","snapshot_observed_at":"2026-08-07T14:16:46.575546Z","title":"H.,Foerster, J.andWhiteson, S.(2019)","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2505.19481","last_updated":"2025-05-26T04:03:48Z","snapshot_observed_at":"2026-08-07T14:11:12.334794Z","submitted_at":"2025-05-26T04:03:48Z","title":"Win Fast or Lose Slow: Balancing Speed and Accuracy in Latency-Sensitive Decisions of LLMs","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-07T14:16:46.575546Z"},"links":{"cited_paper":"/paper/1902.04043","citing_paper":"/paper/2505.19481"},"observation_digest":"sha256:5631f92c8bf46f835bbd66913760e8e54ed9af90f5b54f0510f30d12138c912d","observation_id":"69770212-85d2-4c43-9cde-adfd7935c680","resolution":{"observed_at":"2026-08-07T14:16:46.575546Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2201.03533","last_updated":"2022-10-11T21:30:28Z","snapshot_observed_at":"2026-08-04T19:04:26.422432Z","submitted_at":"2022-01-10T18:47:15Z","title":"SCROLLS: Standardized CompaRison Over Long Language Sequences","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2201.03533","snapshot_observed_at":"2026-08-07T14:16:46.664378Z","title":"et al.(2022)","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2505.19481","last_updated":"2025-05-26T04:03:48Z","snapshot_observed_at":"2026-08-07T14:11:12.334794Z","submitted_at":"2025-05-26T04:03:48Z","title":"Win Fast or Lose Slow: Balancing Speed and Accuracy in Latency-Sensitive Decisions of LLMs","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-07T14:16:46.664378Z"},"links":{"cited_paper":"/paper/2201.03533","citing_paper":"/paper/2505.19481"},"observation_digest":"sha256:9adf86b38854297683b34f2a02e3b092ee6f1a2327c388f41af911ffa9dfc304","observation_id":"eb7d2383-827a-42cf-b567-47861ccc6daf","resolution":{"observed_at":"2026-08-07T14:16:46.664378Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2303.11366","last_updated":"2023-10-10T05:21:45Z","snapshot_observed_at":"2026-07-06T15:05:53.556198Z","submitted_at":"2023-03-20T18:08:50Z","title":"Reflexion: Language Agents with Verbal Reinforcement Learning","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.11366","snapshot_observed_at":"2026-08-07T14:16:46.781573Z","title":"Reflexion: an autonomous agent with dynamic memory and self-reflection.arXiv preprint arXiv:2303.11366,29","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.19481","last_updated":"2025-05-26T04:03:48Z","snapshot_observed_at":"2026-08-07T14:11:12.334794Z","submitted_at":"2025-05-26T04:03:48Z","title":"Win Fast or Lose Slow: Balancing Speed and Accuracy in Latency-Sensitive Decisions of LLMs","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-07T14:16:46.781573Z"},"links":{"cited_paper":"/paper/2303.11366","citing_paper":"/paper/2505.19481"},"observation_digest":"sha256:4d558c93ac7864fff8fe248dd4f112ac8fc17b9b4829719022ef704e7dfacdb4","observation_id":"21bb93ef-e647-4911-8c7d-282f4508b882","resolution":{"observed_at":"2026-08-07T14:16:46.781573Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:16:49.185366Z","title":"M.(2010)","venue":null,"work_id":"1e3a0e75-eedc-4543-865c-858f0299dbd6","year":2010},"citing_paper":{"arxiv_id":"2505.19481","last_updated":"2025-05-26T04:03:48Z","snapshot_observed_at":"2026-08-07T14:11:12.334794Z","submitted_at":"2025-05-26T04:03:48Z","title":"Win Fast or Lose Slow: Balancing Speed and Accuracy in Latency-Sensitive Decisions of LLMs","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-07T14:16:46.915619Z"},"links":{"citing_paper":"/paper/2505.19481"},"observation_digest":"sha256:fefed9355c1a74ad2091bbc6d2ba4236a1b02d3b7874f455675e538d12f18ae2","observation_id":"1bbdd1aa-68d7-42be-bd80-cb373336e275","resolution":{"observed_at":"2026-08-07T14:16:49.255214Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2203.08368","last_updated":"2023-03-05T09:08:51Z","snapshot_observed_at":"2026-08-08T01:31:22.675115Z","submitted_at":"2022-03-16T03:23:50Z","title":"Mixed-Precision Neural Network Quantization via Learned Layer-wise Importance","version":5},"cited_work":{"arxiv_id":"2203.08368","doi":null,"metadata_source":"pith","pith_arxiv_id":"2203.08368","snapshot_observed_at":"2026-08-07T14:16:48.281700Z","title":"Mixed-Precision Neural Network Quantization via Learned Layer-wise Importance","venue":"cs.LG","work_id":"8af8f018-9b75-4807-b59c-1e8dcc5cb6f8","year":2022},"citing_paper":{"arxiv_id":"2505.19481","last_updated":"2025-05-26T04:03:48Z","snapshot_observed_at":"2026-08-07T14:11:12.334794Z","submitted_at":"2025-05-26T04:03:48Z","title":"Win Fast or Lose Slow: Balancing Speed and Accuracy in Latency-Sensitive Decisions of LLMs","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-07T14:16:47.026428Z"},"links":{"cited_paper":"/paper/2203.08368","citing_paper":"/paper/2505.19481"},"observation_digest":"sha256:b55e6a7fd7a594ceb09885ccd388c8f772626c2d959fb916feaa93018d17050c","observation_id":"e0c4bd7e-11da-4a31-8a9d-3046f10ecb7f","resolution":{"observed_at":"2026-08-07T14:16:48.351117Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.19786","last_updated":"2025-03-25T15:52:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-03-25T15:52:34Z","title":"Gemma 3 Technical Report","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.19786","snapshot_observed_at":"2026-08-07T14:16:47.031138Z","title":"Gemma 3 technical report","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.19481","last_updated":"2025-05-26T04:03:48Z","snapshot_observed_at":"2026-08-07T14:11:12.334794Z","submitted_at":"2025-05-26T04:03:48Z","title":"Win Fast or Lose Slow: Balancing Speed and Accuracy in Latency-Sensitive Decisions of LLMs","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-07T14:16:47.031138Z"},"links":{"cited_paper":"/paper/2503.19786","citing_paper":"/paper/2505.19481"},"observation_digest":"sha256:d073691cca1003ed5fcc2549d2303dc20147dab6f756e43e769d0f820f67243b","observation_id":"0f9e06fa-926e-432a-b86a-4d2bedd47ee3","resolution":{"observed_at":"2026-08-07T14:16:47.031138Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2211.10438","last_updated":"2024-03-29T19:21:58Z","snapshot_observed_at":"2026-08-07T09:04:32.994880Z","submitted_at":"2022-11-18T18:59:33Z","title":"SmoothQuant: Accurate and Efficient Post-Training Quantization for Large Language Models","version":7},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2211.10438","snapshot_observed_at":"2026-08-07T14:16:47.052422Z","title":"Smoothquant: Accurate and efficient post-training quantization for large language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.19481","last_updated":"2025-05-26T04:03:48Z","snapshot_observed_at":"2026-08-07T14:11:12.334794Z","submitted_at":"2025-05-26T04:03:48Z","title":"Win Fast or Lose Slow: Balancing Speed and Accuracy in Latency-Sensitive Decisions of LLMs","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-07T14:16:47.052422Z"},"links":{"cited_paper":"/paper/2211.10438","citing_paper":"/paper/2505.19481"},"observation_digest":"sha256:71a3436a1b1bf25967a927ef4bed28319f1196dbffc92fffc6cc644452801fa4","observation_id":"36b0b688-e51f-4466-842c-d5b6b85c391b","resolution":{"observed_at":"2026-08-07T14:16:47.052422Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:16:49.017915Z","title":"J.,Han, X.,Fu, X.,Zhong, T .,Zeng, J.,Song, M","venue":null,"work_id":"9cf3ec51-5f17-4ca6-aeea-1218cad9f7f1","year":2023},"citing_paper":{"arxiv_id":"2505.19481","last_updated":"2025-05-26T04:03:48Z","snapshot_observed_at":"2026-08-07T14:11:12.334794Z","submitted_at":"2025-05-26T04:03:48Z","title":"Win Fast or Lose Slow: Balancing Speed and Accuracy in Latency-Sensitive Decisions of LLMs","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-07T14:16:47.190120Z"},"links":{"citing_paper":"/paper/2505.19481"},"observation_digest":"sha256:3f0c44f23c260a0f119337ff5d4b637bc7b72de899864f422880366d5190f701","observation_id":"3c57875b-06e5-4221-8c4f-c999c258cb9a","resolution":{"observed_at":"2026-08-07T14:16:49.086346Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.03519","last_updated":"2024-11-05T21:54:14Z","snapshot_observed_at":"2026-07-06T19:45:53.289704Z","submitted_at":"2024-11-05T21:54:14Z","title":"AI Metropolis: Scaling Large Language Model-based Multi-Agent Simulation with Out-of-order Execution","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.03519","snapshot_observed_at":"2026-08-07T14:16:47.386502Z","title":"Ai metropolis: Scaling large language model-based multi-agent simulation with out-of-order execution","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.19481","last_updated":"2025-05-26T04:03:48Z","snapshot_observed_at":"2026-08-07T14:11:12.334794Z","submitted_at":"2025-05-26T04:03:48Z","title":"Win Fast or Lose Slow: Balancing Speed and Accuracy in Latency-Sensitive Decisions of LLMs","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-07T14:16:47.386502Z"},"links":{"cited_paper":"/paper/2411.03519","citing_paper":"/paper/2505.19481"},"observation_digest":"sha256:dfdaf403b0bde6019f0dcee0fd230296c95dc32721809e16c4c45bcc4bb5c270","observation_id":"f25ee928-77ca-4228-97cf-f5bf27d343fc","resolution":{"observed_at":"2026-08-07T14:16:47.386502Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-07T14:16:48.886688Z","title":"R.andCao, Y .(2023)","venue":null,"work_id":"f195c84f-e48e-4fb2-8740-a60431686053","year":2023},"citing_paper":{"arxiv_id":"2505.19481","last_updated":"2025-05-26T04:03:48Z","snapshot_observed_at":"2026-08-07T14:11:12.334794Z","submitted_at":"2025-05-26T04:03:48Z","title":"Win Fast or Lose Slow: Balancing Speed and Accuracy in Latency-Sensitive Decisions of LLMs","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-07T14:16:47.528330Z"},"links":{"citing_paper":"/paper/2505.19481"},"observation_digest":"sha256:46e66872977838bac7c9ee19e5d8e633d08c8323692c1a8749920bcfe7ab6d14","observation_id":"fa336cfe-16eb-4ce2-a618-5c0dbaac1711","resolution":{"observed_at":"2026-08-07T14:16:48.928927Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.13743","last_updated":"2023-12-03T16:18:55Z","snapshot_observed_at":"2026-07-06T16:51:23.348409Z","submitted_at":"2023-11-23T00:24:40Z","title":"FinMem: A Performance-Enhanced LLM Trading Agent with Layered Memory and Character Design","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.13743","snapshot_observed_at":"2026-08-07T14:16:47.619235Z","title":"W .andKhashanah, K","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.19481","last_updated":"2025-05-26T04:03:48Z","snapshot_observed_at":"2026-08-07T14:11:12.334794Z","submitted_at":"2025-05-26T04:03:48Z","title":"Win Fast or Lose Slow: Balancing Speed and Accuracy in Latency-Sensitive Decisions of LLMs","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-07T14:16:47.619235Z"},"links":{"cited_paper":"/paper/2311.13743","citing_paper":"/paper/2505.19481"},"observation_digest":"sha256:466b6de18c29ceabfef94ca67610c500d59a249c510835afa2936fb43eb908c0","observation_id":"2d2632fa-ff0d-4efb-8bf8-dc3a53a713a8","resolution":{"observed_at":"2026-08-07T14:16:47.619235Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.18485","last_updated":"2024-06-28T10:35:56Z","snapshot_observed_at":"2026-07-06T17:36:56.979448Z","submitted_at":"2024-02-28T17:06:54Z","title":"A Multimodal Foundation Agent for Financial Trading: Tool-Augmented, Diversified, and Generalist","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.18485","snapshot_observed_at":"2026-08-07T14:16:47.697800Z","title":"A multimodal foundation agent for financial trading: Tool- augmented, diversified, and generalist","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.19481","last_updated":"2025-05-26T04:03:48Z","snapshot_observed_at":"2026-08-07T14:11:12.334794Z","submitted_at":"2025-05-26T04:03:48Z","title":"Win Fast or Lose Slow: Balancing Speed and Accuracy in Latency-Sensitive Decisions of LLMs","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-07T14:16:47.697800Z"},"links":{"cited_paper":"/paper/2402.18485","citing_paper":"/paper/2505.19481"},"observation_digest":"sha256:b7b3a34535df2d48652e599d4f458721e41ac147010180b4a98904c78e316117","observation_id":"5e2ce555-b285-4428-a4e5-b1e7c38b7e90","resolution":{"observed_at":"2026-08-07T14:16:47.697800Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.19102","last_updated":"2024-04-16T06:08:05Z","snapshot_observed_at":"2026-08-05T04:04:11.667434Z","submitted_at":"2023-10-29T18:33:05Z","title":"Atom: Low-bit Quantization for Efficient and Accurate LLM Serving","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.19102","snapshot_observed_at":"2026-08-07T14:16:47.771772Z","title":"andKasikci, B.(2024)","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.19481","last_updated":"2025-05-26T04:03:48Z","snapshot_observed_at":"2026-08-07T14:11:12.334794Z","submitted_at":"2025-05-26T04:03:48Z","title":"Win Fast or Lose Slow: Balancing Speed and Accuracy in Latency-Sensitive Decisions of LLMs","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-07T14:16:47.771772Z"},"links":{"cited_paper":"/paper/2310.19102","citing_paper":"/paper/2505.19481"},"observation_digest":"sha256:11bb0d7b70d30d360775ca2599dc518d7d3d1b57727aef07bf1cae8bfbb13386","observation_id":"da10163d-d38c-4102-ba5e-e01fcc73c35f","resolution":{"observed_at":"2026-08-07T14:16:47.771772Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.07104","last_updated":"2024-06-06T00:10:06Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-12-12T09:34:27Z","title":"SGLang: Efficient Execution of Structured Language Model Programs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.07104","snapshot_observed_at":"2026-08-07T14:16:47.871852Z","title":"H.,Cao, S.,Kozyrakis, C.,Stoica, I.,Gonzalez, J","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.19481","last_updated":"2025-05-26T04:03:48Z","snapshot_observed_at":"2026-08-07T14:11:12.334794Z","submitted_at":"2025-05-26T04:03:48Z","title":"Win Fast or Lose Slow: Balancing Speed and Accuracy in Latency-Sensitive Decisions of LLMs","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-07T14:16:47.871852Z"},"links":{"cited_paper":"/paper/2312.07104","citing_paper":"/paper/2505.19481"},"observation_digest":"sha256:590afd9938711208d82e3043574ed321f1506c6f90eed4817d18bad06a9531e5","observation_id":"5d7a3c2d-ba7d-49ac-908b-6ffb0a82c59f","resolution":{"observed_at":"2026-08-07T14:16:47.871852Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.15877","last_updated":"2025-04-01T08:36:44Z","snapshot_observed_at":"2026-07-31T19:00:59.311189Z","submitted_at":"2024-06-22T15:52:04Z","title":"BigCodeBench: Benchmarking Code Generation with Diverse Function Calls and Complex Instructions","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.15877","snapshot_observed_at":"2026-08-07T14:16:47.986431Z","title":"Y .,Vu, M","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.19481","last_updated":"2025-05-26T04:03:48Z","snapshot_observed_at":"2026-08-07T14:11:12.334794Z","submitted_at":"2025-05-26T04:03:48Z","title":"Win Fast or Lose Slow: Balancing Speed and Accuracy in Latency-Sensitive Decisions of LLMs","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-07T14:16:47.986431Z"},"links":{"cited_paper":"/paper/2406.15877","citing_paper":"/paper/2505.19481"},"observation_digest":"sha256:a6558c4a05706a92dd66ce3a868ffba35b5f34ea6e5ee9d82d20fae5be841c88","observation_id":"5e5b4846-3827-4d00-9ba3-99b763f58db1","resolution":{"observed_at":"2026-08-07T14:16:47.986431Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2505.19481","last_updated":"2025-05-26T04:03:48Z","latest_version":1,"primary_category":"cs.LG","snapshot_observed_at":"2026-08-07T14:11:12.334794Z","submitted_at":"2025-05-26T04:03:48Z","title":"Win Fast or Lose Slow: Balancing Speed and Accuracy in Latency-Sensitive Decisions of LLMs"},"reference_resolution":{"displayed":36,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":25,"verified_exact":2,"verified_fuzzy":9},"total_outbound_references":36},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"thesis":"As of 8 August 2026, this Paper Citation Record lists 36 of 36 outbound references and 1 inbound Pith citation observation for arXiv:2505.19481."}