{"as_of":"2026-08-18T05:57:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:fade4f8fae491565694ddd9d650d9b529e01e94746636439f87188882b0e8be3","coverage":[{"denominator":52,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":52,"source":"paper_references, paper_reference_links","source_observed_at":"2026-05-16T21:57:12.142867Z","state":"measured"},{"denominator":53,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":53,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-17T06:30:58.91139+00:00","state":"measured"},{"denominator":1,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":1,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-02T05:30:33.693227Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2512.16056","last_updated":"2026-05-13T15:33:38Z","snapshot_observed_at":"2026-08-15T04:35:21.537340Z","submitted_at":"2025-12-18T00:45:00Z","title":"MultiPath Memory Access: Breaking Host-GPU Bandwidth Bottlenecks in LLM Services","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2512.16056","snapshot_observed_at":"2026-08-02T05:30:33.693227Z","title":"arXiv:2512.16056 [cs.DC] https://arxiv.org/abs/2512.16056","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.13352","last_updated":"2026-07-15T00:24:01Z","snapshot_observed_at":"2026-08-14T01:37:27.287153Z","submitted_at":"2026-07-15T00:24:01Z","title":"HybridQC: Hardware-Grounded Simulation of Tightly Integrated Hybrid Quantum-Classical Systems","version":1},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-02T05:30:33.693227Z"},"links":{"cited_paper":"/paper/2512.16056","citing_paper":"/paper/2607.13352"},"observation_digest":"sha256:63e51c59648af7ed87663c56f028065097b0208131c250ca811f0911995eca72","observation_id":"718b15f1-63ac-4ccd-af80-703bb99b4d5b","resolution":{"observed_at":"2026-08-02T05:30:33.693227Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2512.16056/citation-record","integrity":"/paper/2512.16056/integrity","json":"/paper/2512.16056/citation-record.json","paper":"/paper/2512.16056"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"62fb66e6-930b-43a1-be90-57ee36b0ba4a","year":null},"citing_paper":{"arxiv_id":"2512.16056","last_updated":"2026-05-13T15:33:38Z","snapshot_observed_at":"2026-08-15T04:35:21.537340Z","submitted_at":"2025-12-18T00:45:00Z","title":"MultiPath Memory Access: Breaking Host-GPU Bandwidth Bottlenecks in LLM Services","version":2},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-16T21:57:12.142867Z"},"links":{"citing_paper":"/paper/2512.16056"},"observation_digest":"sha256:f9631e24e61ccf4ece9f502e96127245f158c4e439c6c0a14217c819c0f63b31","observation_id":"a1512afd-1f2e-4827-a3ab-bb35a7d8634a","resolution":{"observed_at":"2026-05-16T21:58:36.486195Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Advanced Micro Devices","venue":null,"work_id":"9be9b912-df33-4373-a31f-11476d1f55b2","year":2022},"citing_paper":{"arxiv_id":"2512.16056","last_updated":"2026-05-13T15:33:38Z","snapshot_observed_at":"2026-08-15T04:35:21.537340Z","submitted_at":"2025-12-18T00:45:00Z","title":"MultiPath Memory Access: Breaking Host-GPU Bandwidth Bottlenecks in LLM Services","version":2},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-16T21:57:12.142867Z"},"links":{"citing_paper":"/paper/2512.16056"},"observation_digest":"sha256:931a22ccf94cb10127a5b4f35767a471fae0f255da502fbea69acf66030ea90c","observation_id":"48d2025b-c6e0-404a-ae87-50c2b4b82577","resolution":{"observed_at":"2026-05-16T21:58:36.488386Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Advanced Micro Devices","venue":null,"work_id":"89774536-8edc-4183-ac14-ed01cf54940b","year":2024},"citing_paper":{"arxiv_id":"2512.16056","last_updated":"2026-05-13T15:33:38Z","snapshot_observed_at":"2026-08-15T04:35:21.537340Z","submitted_at":"2025-12-18T00:45:00Z","title":"MultiPath Memory Access: Breaking Host-GPU Bandwidth Bottlenecks in LLM Services","version":2},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-16T21:57:12.142867Z"},"links":{"citing_paper":"/paper/2512.16056"},"observation_digest":"sha256:8a50181755f7529ce028800835c625d1fa8b244171b4cdf41ed397e47b225f7e","observation_id":"1d7481eb-1776-4adb-bc86-40358fbfdbea","resolution":{"observed_at":"2026-05-16T21:58:36.493061Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Deepspeed-inference: enabling efficient in- ference of transformer models at unprecedented scale","venue":null,"work_id":"461bf7aa-e1b9-4309-a540-e2905d823c31","year":2022},"citing_paper":{"arxiv_id":"2512.16056","last_updated":"2026-05-13T15:33:38Z","snapshot_observed_at":"2026-08-15T04:35:21.537340Z","submitted_at":"2025-12-18T00:45:00Z","title":"MultiPath Memory Access: Breaking Host-GPU Bandwidth Bottlenecks in LLM Services","version":2},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-16T21:57:12.142867Z"},"links":{"citing_paper":"/paper/2512.16056"},"observation_digest":"sha256:1bdce3ddb5d75bf8ce5cf6c224717c087443265c84f2ba3d8b06e0da38c7cc72","observation_id":"e5f7dd5e-13f7-4efa-9215-0114396729dc","resolution":{"observed_at":"2026-05-16T21:58:36.490728Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2309.16609","last_updated":"2023-09-28T17:07:49Z","snapshot_observed_at":"2026-08-09T21:25:20.369782Z","submitted_at":"2023-09-28T17:07:49Z","title":"Qwen Technical Report","version":1},"cited_work":{"arxiv_id":"2309.16609","doi":"10.48550/arxiv.2309.16609","metadata_source":"pith","pith_arxiv_id":"2309.16609","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Qwen Technical Report","venue":"cs.CL","work_id":"bb1fd52f-6b2f-437c-9516-37bdf6eb9be8","year":2023},"citing_paper":{"arxiv_id":"2512.16056","last_updated":"2026-05-13T15:33:38Z","snapshot_observed_at":"2026-08-15T04:35:21.537340Z","submitted_at":"2025-12-18T00:45:00Z","title":"MultiPath Memory Access: Breaking Host-GPU Bandwidth Bottlenecks in LLM Services","version":2},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-05-16T21:57:12.142867Z"},"links":{"cited_paper":"/paper/2309.16609","citing_paper":"/paper/2512.16056"},"observation_digest":"sha256:ddc9dea0f198b3fbbe92d87d7ff5e78fe5fbea08890d9be0e0f31aa108aecb37","observation_id":"c2cf82dc-bc87-4c6e-a579-2554c3aee017","resolution":{"observed_at":"2026-05-16T21:58:35.765384Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T16:38:14.101106+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T16:38:14.101106+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-17T16:38:14.101106+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.15204","last_updated":"2025-01-03T11:44:51Z","snapshot_observed_at":"2026-08-12T12:34:50.226758Z","submitted_at":"2024-12-19T18:59:17Z","title":"LongBench v2: Towards Deeper Understanding and Reasoning on Realistic Long-context Multitasks","version":2},"cited_work":{"arxiv_id":"2412.15204","doi":"10.48550/arxiv.2412.15204","metadata_source":"pith","pith_arxiv_id":"2412.15204","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"LongBench v2: Towards Deeper Understanding and Reasoning on Realistic Long-context Multitasks","venue":"cs.CL","work_id":"9fac250b-241e-41ce-9177-469deaf03040","year":2024},"citing_paper":{"arxiv_id":"2512.16056","last_updated":"2026-05-13T15:33:38Z","snapshot_observed_at":"2026-08-15T04:35:21.537340Z","submitted_at":"2025-12-18T00:45:00Z","title":"MultiPath Memory Access: Breaking Host-GPU Bandwidth Bottlenecks in LLM Services","version":2},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-05-16T21:57:12.142867Z"},"links":{"cited_paper":"/paper/2412.15204","citing_paper":"/paper/2512.16056"},"observation_digest":"sha256:179b96d2e52c4ac6f62540b4723f1c43ce0fa5341439c3b0bed5c5f5bdd625fc","observation_id":"7f3bb36f-358e-44ec-94d2-3d7d9f0e9c42","resolution":{"observed_at":"2026-05-16T21:58:35.743753Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"{PipeSwitch}: Fast pipelined context switching for deep learning applications","venue":null,"work_id":"c837ab2b-cff8-4e56-ad3f-1bd8c80f888c","year":2020},"citing_paper":{"arxiv_id":"2512.16056","last_updated":"2026-05-13T15:33:38Z","snapshot_observed_at":"2026-08-15T04:35:21.537340Z","submitted_at":"2025-12-18T00:45:00Z","title":"MultiPath Memory Access: Breaking Host-GPU Bandwidth Bottlenecks in LLM Services","version":2},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-05-16T21:57:12.142867Z"},"links":{"citing_paper":"/paper/2512.16056"},"observation_digest":"sha256:be9810614066a22d5bbb68470594a7e8941c7ec0f51182a77080dde1b791c39b","observation_id":"310ba4a2-ed39-4f40-b360-86edd3eb2c6e","resolution":{"observed_at":"2026-05-16T21:58:36.529197Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Overlapping data transfers with computation on gpu with tiles","venue":null,"work_id":"1cf89701-6cc9-4567-9375-1c13b26360d9","year":2017},"citing_paper":{"arxiv_id":"2512.16056","last_updated":"2026-05-13T15:33:38Z","snapshot_observed_at":"2026-08-15T04:35:21.537340Z","submitted_at":"2025-12-18T00:45:00Z","title":"MultiPath Memory Access: Breaking Host-GPU Bandwidth Bottlenecks in LLM Services","version":2},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-05-16T21:57:12.142867Z"},"links":{"citing_paper":"/paper/2512.16056"},"observation_digest":"sha256:9f996354cfb705796ccab9b71428d626e79f4dfd039ea1439f19a354c6d32efe","observation_id":"eb15c589-e7a2-4955-9b6d-3439f5f42411","resolution":{"observed_at":"2026-05-16T21:58:36.531268Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"NVIDIA H20 GPU Specifications","venue":null,"work_id":"02e12822-b586-407a-81f4-b5e73284cc9f","year":2024},"citing_paper":{"arxiv_id":"2512.16056","last_updated":"2026-05-13T15:33:38Z","snapshot_observed_at":"2026-08-15T04:35:21.537340Z","submitted_at":"2025-12-18T00:45:00Z","title":"MultiPath Memory Access: Breaking Host-GPU Bandwidth Bottlenecks in LLM Services","version":2},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-05-16T21:57:12.142867Z"},"links":{"citing_paper":"/paper/2512.16056"},"observation_digest":"sha256:fdbf788dc1c31a0b8ba281d0c64769e87f5cfe80282d82d5a880791200515acf","observation_id":"ddb17191-178f-42d2-b8a3-9868ae585bef","resolution":{"observed_at":"2026-05-16T21:58:36.535412Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"MP-RDMA: enabling RDMA with multi- path transport in datacenters.IEEE/ACM Trans","venue":null,"work_id":"0fcd50c3-8d02-4e1b-acc9-fd89b3e1bd61","year":2019},"citing_paper":{"arxiv_id":"2512.16056","last_updated":"2026-05-13T15:33:38Z","snapshot_observed_at":"2026-08-15T04:35:21.537340Z","submitted_at":"2025-12-18T00:45:00Z","title":"MultiPath Memory Access: Breaking Host-GPU Bandwidth Bottlenecks in LLM Services","version":2},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-05-16T21:57:12.142867Z"},"links":{"citing_paper":"/paper/2512.16056"},"observation_digest":"sha256:e93cdb233e0154884dae86a9cdc538a2d7125dbc70bb78d23ab8b1c6da057d7f","observation_id":"e07bb5a2-e984-4ace-ac87-c84fa74f5b1e","resolution":{"observed_at":"2026-05-16T21:58:36.539868Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2510.09665","doi":"10.48550/arxiv.2510.09665","metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Lmcache: An efficient kv cache layer for enterprise-scale llm inference","venue":"arXiv (Cornell University)","work_id":"089b937e-f3f9-4525-a792-524ef0f7db1d","year":2025},"citing_paper":{"arxiv_id":"2512.16056","last_updated":"2026-05-13T15:33:38Z","snapshot_observed_at":"2026-08-15T04:35:21.537340Z","submitted_at":"2025-12-18T00:45:00Z","title":"MultiPath Memory Access: Breaking Host-GPU Bandwidth Bottlenecks in LLM Services","version":2},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-05-16T21:57:12.142867Z"},"links":{"citing_paper":"/paper/2512.16056"},"observation_digest":"sha256:eb0554f299c1deb6f906366f5f5c5eaac2eeae9a6cee62d4aa30912cc72e62c8","observation_id":"441cb060-0925-4e7f-87fa-a17e4508c910","resolution":{"observed_at":"2026-05-16T21:58:35.747666Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2507.14397","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-11T01:57:51.083148Z","title":"Liminal: Exploring the frontiers of llm decode performance","venue":null,"work_id":"23f71827-7a8a-414d-9118-39d156485729","year":2025},"citing_paper":{"arxiv_id":"2512.16056","last_updated":"2026-05-13T15:33:38Z","snapshot_observed_at":"2026-08-15T04:35:21.537340Z","submitted_at":"2025-12-18T00:45:00Z","title":"MultiPath Memory Access: Breaking Host-GPU Bandwidth Bottlenecks in LLM Services","version":2},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-05-16T21:57:12.142867Z"},"links":{"citing_paper":"/paper/2512.16056"},"observation_digest":"sha256:93dc83e6ec111038c62a82cc33ac02cac2f73ada973ce2aae4d1ecf3d93f9ce0","observation_id":"371f2c2c-782a-4ef7-9724-bc68ec9dde9b","resolution":{"observed_at":"2026-05-16T21:58:35.736890Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.04594","last_updated":"2025-05-23T08:34:50Z","snapshot_observed_at":"2026-08-16T13:45:18.351922Z","submitted_at":"2024-06-07T02:58:35Z","title":"Enhancing Large-Scale AI Training Efficiency: The C4 Solution for Real-Time Anomaly Detection and Communication Optimization","version":2},"cited_work":{"arxiv_id":"2406.04594","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2406.04594","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Boosting large-scale parallel training efficiency with c4: A communication-driven approach","venue":null,"work_id":"a7a73121-fb7d-4be3-a9fa-53b60507e0ab","year":2024},"citing_paper":{"arxiv_id":"2512.16056","last_updated":"2026-05-13T15:33:38Z","snapshot_observed_at":"2026-08-15T04:35:21.537340Z","submitted_at":"2025-12-18T00:45:00Z","title":"MultiPath Memory Access: Breaking Host-GPU Bandwidth Bottlenecks in LLM Services","version":2},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-05-16T21:57:12.142867Z"},"links":{"cited_paper":"/paper/2406.04594","citing_paper":"/paper/2512.16056"},"observation_digest":"sha256:6f6a026b43328a8f459c917907c61725f81e34158c8ef0a268f38404e5930c94","observation_id":"cceb7542-ddcf-4471-bc72-697c3d9c0b79","resolution":{"observed_at":"2026-05-16T21:58:35.762214Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"{Cost-Efficient} large lan- guage model serving for multi-turn conversations with {CachedAttention}","venue":null,"work_id":"884ae2a9-cfe9-40fb-a19a-feee7dcd9603","year":2024},"citing_paper":{"arxiv_id":"2512.16056","last_updated":"2026-05-13T15:33:38Z","snapshot_observed_at":"2026-08-15T04:35:21.537340Z","submitted_at":"2025-12-18T00:45:00Z","title":"MultiPath Memory Access: Breaking Host-GPU Bandwidth Bottlenecks in LLM Services","version":2},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-05-16T21:57:12.142867Z"},"links":{"citing_paper":"/paper/2512.16056"},"observation_digest":"sha256:b0bcdbe57dfe3394ec7228e3614072b109b115d7e957083b9bea9606ed8a5d27","observation_id":"8d1506e8-3b68-4bd1-98c5-17b93ee2a7a5","resolution":{"observed_at":"2026-05-16T21:58:36.495681Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Prompt cache: Modular attention reuse for low-latency inference.Pro- ceedings of Machine Learning and Systems, 6:325–338","venue":null,"work_id":"148eb982-54e7-4722-89b0-84bcc236e13c","year":2024},"citing_paper":{"arxiv_id":"2512.16056","last_updated":"2026-05-13T15:33:38Z","snapshot_observed_at":"2026-08-15T04:35:21.537340Z","submitted_at":"2025-12-18T00:45:00Z","title":"MultiPath Memory Access: Breaking Host-GPU Bandwidth Bottlenecks in LLM Services","version":2},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-05-16T21:57:12.142867Z"},"links":{"citing_paper":"/paper/2512.16056"},"observation_digest":"sha256:e0083dc43cc418d498229abd16454c2634194370389f18d70c8d5d33dd25f10c","observation_id":"e056e66c-bdf6-4623-9a8e-ecae4a64f8b1","resolution":{"observed_at":"2026-05-16T21:58:36.584662Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Accelerate: Train- ing and inference at scale made simple, efficient and adaptable","venue":null,"work_id":"58f9e652-0345-422b-92f9-f8b6f6f70cd9","year":2022},"citing_paper":{"arxiv_id":"2512.16056","last_updated":"2026-05-13T15:33:38Z","snapshot_observed_at":"2026-08-15T04:35:21.537340Z","submitted_at":"2025-12-18T00:45:00Z","title":"MultiPath Memory Access: Breaking Host-GPU Bandwidth Bottlenecks in LLM Services","version":2},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-05-16T21:57:12.142867Z"},"links":{"citing_paper":"/paper/2512.16056"},"observation_digest":"sha256:879edb9b67587414fc6adaa5ae1f1718be96ac3e6b8b6f4859e8a02cf2d633b3","observation_id":"c6646b8f-7bd8-46ac-80fc-b3f47c3709aa","resolution":{"observed_at":"2026-05-16T21:58:36.586818Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Elsevier","venue":null,"work_id":"c336b932-fc23-4ac2-b545-17ef3a8db6ac","year":2011},"citing_paper":{"arxiv_id":"2512.16056","last_updated":"2026-05-13T15:33:38Z","snapshot_observed_at":"2026-08-15T04:35:21.537340Z","submitted_at":"2025-12-18T00:45:00Z","title":"MultiPath Memory Access: Breaking Host-GPU Bandwidth Bottlenecks in LLM Services","version":2},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-05-16T21:57:12.142867Z"},"links":{"citing_paper":"/paper/2512.16056"},"observation_digest":"sha256:f0458a3357ce49fd6f7c7ac24bc8dd7c62bbc92a1c320eac01c0d75e4b01273b","observation_id":"026b2508-a03a-44f9-92d9-3f63239a9459","resolution":{"observed_at":"2026-05-16T21:58:36.578185Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"In 20th USENIX Symposium on Networked Systems Design and Implementation (NSDI 23), pages 87–101","venue":null,"work_id":"3fe28222-b064-43f5-bee8-78ad7d7c2ebf","year":2023},"citing_paper":{"arxiv_id":"2512.16056","last_updated":"2026-05-13T15:33:38Z","snapshot_observed_at":"2026-08-15T04:35:21.537340Z","submitted_at":"2025-12-18T00:45:00Z","title":"MultiPath Memory Access: Breaking Host-GPU Bandwidth Bottlenecks in LLM Services","version":2},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-05-16T21:57:12.142867Z"},"links":{"citing_paper":"/paper/2512.16056"},"observation_digest":"sha256:34d2b0ed8251596abd4257e1f41eb5593e2c214661b9dee8e74307fd1e1adeee","observation_id":"a2db5fe7-fac7-4534-a938-4d2fd0429f50","resolution":{"observed_at":"2026-05-16T21:58:36.582402Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Ragcache: Efficient knowledge caching for retrieval-augmented generation","venue":null,"work_id":"4fa8bc24-37a8-4231-b207-dbb3d16d4786","year":2024},"citing_paper":{"arxiv_id":"2512.16056","last_updated":"2026-05-13T15:33:38Z","snapshot_observed_at":"2026-08-15T04:35:21.537340Z","submitted_at":"2025-12-18T00:45:00Z","title":"MultiPath Memory Access: Breaking Host-GPU Bandwidth Bottlenecks in LLM Services","version":2},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-05-16T21:57:12.142867Z"},"links":{"citing_paper":"/paper/2512.16056"},"observation_digest":"sha256:debf367ef4836cab8b0f14ea6b7d0744473aaef99186be6f9de4beed1b674b8c","observation_id":"32fd17b9-fcf1-4339-ae4d-b7db99498a3a","resolution":{"observed_at":"2026-05-16T21:58:36.569355Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"In- put/output memory management unit with protection mode for preventing memory access by i/o devices, Jan- uary 14 2014","venue":null,"work_id":"c7d13690-3619-42ed-a9cc-d866465edde8","year":2014},"citing_paper":{"arxiv_id":"2512.16056","last_updated":"2026-05-13T15:33:38Z","snapshot_observed_at":"2026-08-15T04:35:21.537340Z","submitted_at":"2025-12-18T00:45:00Z","title":"MultiPath Memory Access: Breaking Host-GPU Bandwidth Bottlenecks in LLM Services","version":2},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-05-16T21:57:12.142867Z"},"links":{"citing_paper":"/paper/2512.16056"},"observation_digest":"sha256:4db0e8c7a4b04c01aad89f925f53f96cf1e701b2f34ed1564b3480f72484ff77","observation_id":"70f74dba-9826-46ee-aa2c-81c5d008f5c6","resolution":{"observed_at":"2026-05-16T21:58:36.564948Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Efficient memory manage- ment for large language model serving with pagedatten- tion","venue":null,"work_id":"1838faa3-6cc7-4ab9-83cf-7fa4b028d6c4","year":2023},"citing_paper":{"arxiv_id":"2512.16056","last_updated":"2026-05-13T15:33:38Z","snapshot_observed_at":"2026-08-15T04:35:21.537340Z","submitted_at":"2025-12-18T00:45:00Z","title":"MultiPath Memory Access: Breaking Host-GPU Bandwidth Bottlenecks in LLM Services","version":2},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-05-16T21:57:12.142867Z"},"links":{"citing_paper":"/paper/2512.16056"},"observation_digest":"sha256:c78bbaebf5727f1d9bbc83d7ec121e0f96ea32dbcc387abf89c499a811a732c7","observation_id":"dafb295d-15b6-44eb-a207-961b9c88ad66","resolution":{"observed_at":"2026-05-16T21:58:36.571780Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Tuccl: Tailored and unified configuration optimizations for high-performance collective communication library","venue":null,"work_id":"71ef53f1-bc96-4a98-b0ea-45e92440eabf","year":2025},"citing_paper":{"arxiv_id":"2512.16056","last_updated":"2026-05-13T15:33:38Z","snapshot_observed_at":"2026-08-15T04:35:21.537340Z","submitted_at":"2025-12-18T00:45:00Z","title":"MultiPath Memory Access: Breaking Host-GPU Bandwidth Bottlenecks in LLM Services","version":2},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-05-16T21:57:12.142867Z"},"links":{"citing_paper":"/paper/2512.16056"},"observation_digest":"sha256:70d27daeb2095dc40bced5e1ef2d464fba50959a106c5f1b2141424bdf900791","observation_id":"9e23028c-9366-4374-80a2-e0e627c92fd4","resolution":{"observed_at":"2026-05-16T21:58:36.562583Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Reducing gpu offload latency via fine-grained cpu-gpu synchroniza- tion","venue":null,"work_id":"0cfc53f7-95df-43a2-abde-2e9c84b6775e","year":2013},"citing_paper":{"arxiv_id":"2512.16056","last_updated":"2026-05-13T15:33:38Z","snapshot_observed_at":"2026-08-15T04:35:21.537340Z","submitted_at":"2025-12-18T00:45:00Z","title":"MultiPath Memory Access: Breaking Host-GPU Bandwidth Bottlenecks in LLM Services","version":2},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-05-16T21:57:12.142867Z"},"links":{"citing_paper":"/paper/2512.16056"},"observation_digest":"sha256:613e287904f9505650c43f7456b5c191bdb1fa8d2ca7d81e2cad3a8bb28084ef","observation_id":"a2b88dfe-c9fe-4015-839c-682ca6669fa6","resolution":{"observed_at":"2026-05-16T21:58:36.555180Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Mlp-offload: Multi- level, multi-path offloading for llm pre-training to break the gpu memory wall","venue":null,"work_id":"322d350a-52ff-490b-b850-cdaf6a890398","year":2025},"citing_paper":{"arxiv_id":"2512.16056","last_updated":"2026-05-13T15:33:38Z","snapshot_observed_at":"2026-08-15T04:35:21.537340Z","submitted_at":"2025-12-18T00:45:00Z","title":"MultiPath Memory Access: Breaking Host-GPU Bandwidth Bottlenecks in LLM Services","version":2},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-05-16T21:57:12.142867Z"},"links":{"citing_paper":"/paper/2512.16056"},"observation_digest":"sha256:54b4d9b1abd988e548dda5e7f6cfc8b7fbf37b57f0f1bdd439807e3548bd65b0","observation_id":"79106038-1976-4d4b-8dc4-aeeff2c198aa","resolution":{"observed_at":"2026-05-16T21:58:36.557553Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"AzurePublicDataset: Azure LLM In- ference Dataset 2023","venue":null,"work_id":"272b374f-e2c5-4055-b05f-0b4d6349592b","year":2023},"citing_paper":{"arxiv_id":"2512.16056","last_updated":"2026-05-13T15:33:38Z","snapshot_observed_at":"2026-08-15T04:35:21.537340Z","submitted_at":"2025-12-18T00:45:00Z","title":"MultiPath Memory Access: Breaking Host-GPU Bandwidth Bottlenecks in LLM Services","version":2},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-05-16T21:57:12.142867Z"},"links":{"citing_paper":"/paper/2512.16056"},"observation_digest":"sha256:da590facdeaefbba75503b43793de1134ad08678212ab7f1833ec77b375c2da9","observation_id":"c99002f3-e5c4-4b3a-8ed7-5176e83740b5","resolution":{"observed_at":"2026-05-16T21:58:36.559940Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Scalable parallel programming with cuda: Is cuda the parallel programming model that application developers have been waiting for?Queue, 6(2):40–53","venue":null,"work_id":"dcc7176f-ec43-40e7-9d1a-f31a1ea5e41f","year":2008},"citing_paper":{"arxiv_id":"2512.16056","last_updated":"2026-05-13T15:33:38Z","snapshot_observed_at":"2026-08-15T04:35:21.537340Z","submitted_at":"2025-12-18T00:45:00Z","title":"MultiPath Memory Access: Breaking Host-GPU Bandwidth Bottlenecks in LLM Services","version":2},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-05-16T21:57:12.142867Z"},"links":{"citing_paper":"/paper/2512.16056"},"observation_digest":"sha256:c48fb4f9c2738153b835af4f862ae8de1f4e5626419ad662e6f4bf9652c6a01b","observation_id":"cce9ab10-0984-43d3-b0a8-5f1031e7d7b7","resolution":{"observed_at":"2026-05-16T21:58:36.567284Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"NVLink and NVSwitch","venue":null,"work_id":"0a498de1-32ce-4f33-a3ab-0f87c9eecb3b","year":2024},"citing_paper":{"arxiv_id":"2512.16056","last_updated":"2026-05-13T15:33:38Z","snapshot_observed_at":"2026-08-15T04:35:21.537340Z","submitted_at":"2025-12-18T00:45:00Z","title":"MultiPath Memory Access: Breaking Host-GPU Bandwidth Bottlenecks in LLM Services","version":2},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-05-16T21:57:12.142867Z"},"links":{"citing_paper":"/paper/2512.16056"},"observation_digest":"sha256:2bba3078475942858fd80877b55d7b98af27ec6d2bc7d2423f8ac4885f99c945","observation_id":"41890653-6936-48b1-9be6-f801bb00176f","resolution":{"observed_at":"2026-05-16T21:58:36.580068Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"NVIDIA NVLink 4.0 Technology","venue":null,"work_id":"a7285591-522b-4bf8-89da-36c4260247cb","year":2025},"citing_paper":{"arxiv_id":"2512.16056","last_updated":"2026-05-13T15:33:38Z","snapshot_observed_at":"2026-08-15T04:35:21.537340Z","submitted_at":"2025-12-18T00:45:00Z","title":"MultiPath Memory Access: Breaking Host-GPU Bandwidth Bottlenecks in LLM Services","version":2},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-05-16T21:57:12.142867Z"},"links":{"citing_paper":"/paper/2512.16056"},"observation_digest":"sha256:730573b1c4294d87f3df368d5e29c14ad5c209774edc4f4ee8183e47c271dd64","observation_id":"9a3dfa21-96ed-430f-9a82-8c784f89f533","resolution":{"observed_at":"2026-05-16T21:58:36.550031Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Chatbot Arena Conversations Dataset","venue":null,"work_id":"f575523f-8a3c-4faf-86de-a544287c3fe6","year":2023},"citing_paper":{"arxiv_id":"2512.16056","last_updated":"2026-05-13T15:33:38Z","snapshot_observed_at":"2026-08-15T04:35:21.537340Z","submitted_at":"2025-12-18T00:45:00Z","title":"MultiPath Memory Access: Breaking Host-GPU Bandwidth Bottlenecks in LLM Services","version":2},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-05-16T21:57:12.142867Z"},"links":{"citing_paper":"/paper/2512.16056"},"observation_digest":"sha256:44976d014cab0c506c636599e9b4395d41867da6d2955377df036924122504bb","observation_id":"162c1464-a930-465d-b04d-edc251fde8d1","resolution":{"observed_at":"2026-05-16T21:58:36.547646Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.19379","last_updated":"2025-04-10T05:06:29Z","snapshot_observed_at":"2026-08-16T06:11:44.384590Z","submitted_at":"2024-11-28T21:10:20Z","title":"Marconi: Prefix Caching for the Era of Hybrid LLMs","version":3},"cited_work":{"arxiv_id":"2411.19379","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2411.19379","snapshot_observed_at":"2026-07-04T07:29:39.044039Z","title":"Marconi: Prefix caching for the era of hybrid llms","venue":null,"work_id":"03b57d6a-20bc-4aa0-96d6-67aaf18b6a1f","year":2024},"citing_paper":{"arxiv_id":"2512.16056","last_updated":"2026-05-13T15:33:38Z","snapshot_observed_at":"2026-08-15T04:35:21.537340Z","submitted_at":"2025-12-18T00:45:00Z","title":"MultiPath Memory Access: Breaking Host-GPU Bandwidth Bottlenecks in LLM Services","version":2},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-05-16T21:57:12.142867Z"},"links":{"cited_paper":"/paper/2411.19379","citing_paper":"/paper/2512.16056"},"observation_digest":"sha256:671d4464c93454d37b6a12b83dd579649e64acc89da26eb64852e5b25d0b2576","observation_id":"843df7a1-6ebe-4d5e-8af9-c9770fa13f3c","resolution":{"observed_at":"2026-05-16T21:58:35.758757Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Pci express® base specification revision 5.0 version 1.0","venue":null,"work_id":"25eefecc-36ea-4af3-b470-1a6944ac0792","year":2019},"citing_paper":{"arxiv_id":"2512.16056","last_updated":"2026-05-13T15:33:38Z","snapshot_observed_at":"2026-08-15T04:35:21.537340Z","submitted_at":"2025-12-18T00:45:00Z","title":"MultiPath Memory Access: Breaking Host-GPU Bandwidth Bottlenecks in LLM Services","version":2},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-05-16T21:57:12.142867Z"},"links":{"citing_paper":"/paper/2512.16056"},"observation_digest":"sha256:e83e37a12c95e3f09fd44324c85845e8200e2b05d718bd049bd5a3499e24b65c","observation_id":"6b2507e1-9e05-41dd-854a-cd7eaa912f0a","resolution":{"observed_at":"2026-05-16T21:58:36.533391Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Zero-infinity: Breaking the gpu memory wall for extreme scale deep learning","venue":null,"work_id":"55756488-85e2-4245-b7a4-77ad218f3d80","year":2021},"citing_paper":{"arxiv_id":"2512.16056","last_updated":"2026-05-13T15:33:38Z","snapshot_observed_at":"2026-08-15T04:35:21.537340Z","submitted_at":"2025-12-18T00:45:00Z","title":"MultiPath Memory Access: Breaking Host-GPU Bandwidth Bottlenecks in LLM Services","version":2},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-05-16T21:57:12.142867Z"},"links":{"citing_paper":"/paper/2512.16056"},"observation_digest":"sha256:5b18efc73acabeb6f3aefc9a5347ed2d73181993d4f4fea6044850da56d86ac0","observation_id":"4bc0f6bd-9623-4d58-a1c9-3122aabcdcf4","resolution":{"observed_at":"2026-05-16T21:58:36.537501Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"An i/o characterizing study of offloading llm models and kv caches to nvme ssd","venue":null,"work_id":"f5ece792-06d7-48ac-bd9e-e7981efdec1e","year":2025},"citing_paper":{"arxiv_id":"2512.16056","last_updated":"2026-05-13T15:33:38Z","snapshot_observed_at":"2026-08-15T04:35:21.537340Z","submitted_at":"2025-12-18T00:45:00Z","title":"MultiPath Memory Access: Breaking Host-GPU Bandwidth Bottlenecks in LLM Services","version":2},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-05-16T21:57:12.142867Z"},"links":{"citing_paper":"/paper/2512.16056"},"observation_digest":"sha256:51b5521d603f7d55ae68e3e3554b9f27911e4988d369d236b22a726250667d66","observation_id":"8266e2cf-d829-4b34-a41e-2a95666e08ed","resolution":{"observed_at":"2026-05-16T21:58:36.552527Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Enabling efficient GPU communication over mul- tiple NICs with fuselink","venue":null,"work_id":"657d5e67-46b6-4250-a3b4-c7566b8331c6","year":2025},"citing_paper":{"arxiv_id":"2512.16056","last_updated":"2026-05-13T15:33:38Z","snapshot_observed_at":"2026-08-15T04:35:21.537340Z","submitted_at":"2025-12-18T00:45:00Z","title":"MultiPath Memory Access: Breaking Host-GPU Bandwidth Bottlenecks in LLM Services","version":2},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-05-16T21:57:12.142867Z"},"links":{"citing_paper":"/paper/2512.16056"},"observation_digest":"sha256:93b26423a9b9884b3bfeef662875e1d2654578026faf9ffab73930549fe60e13","observation_id":"2a81e73d-0f78-4d56-99f9-bbb1da58781d","resolution":{"observed_at":"2026-05-16T21:58:36.574062Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.04985","last_updated":"2024-09-04T10:04:52Z","snapshot_observed_at":"2026-08-16T14:36:33.398897Z","submitted_at":"2023-12-08T11:47:35Z","title":"SparQ Attention: Bandwidth-Efficient LLM Inference","version":6},"cited_work":{"arxiv_id":"2312.04985","doi":"10.48550/arxiv.2312.04985","metadata_source":"arxiv_reference","pith_arxiv_id":"2312.04985","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Sparq attention: Bandwidth-efficient llm inference","venue":"arXiv (Cornell University)","work_id":"1fa58820-f3b9-446e-9b1d-5b9ab21405a2","year":2023},"citing_paper":{"arxiv_id":"2512.16056","last_updated":"2026-05-13T15:33:38Z","snapshot_observed_at":"2026-08-15T04:35:21.537340Z","submitted_at":"2025-12-18T00:45:00Z","title":"MultiPath Memory Access: Breaking Host-GPU Bandwidth Bottlenecks in LLM Services","version":2},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-05-16T21:57:12.142867Z"},"links":{"cited_paper":"/paper/2312.04985","citing_paper":"/paper/2512.16056"},"observation_digest":"sha256:855b4bdf828a73b06defc6d6adc9cc726cd6d208b0e2a833c9bdcff401db903e","observation_id":"a5967116-cafb-43bb-bcb4-87eacc2ed466","resolution":{"observed_at":"2026-05-16T21:58:35.755161Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Flexgen: high-throughput generative inference of large language models with a single gpu","venue":null,"work_id":"b5bbd58d-46a9-4118-90db-29f36d88c062","year":2023},"citing_paper":{"arxiv_id":"2512.16056","last_updated":"2026-05-13T15:33:38Z","snapshot_observed_at":"2026-08-15T04:35:21.537340Z","submitted_at":"2025-12-18T00:45:00Z","title":"MultiPath Memory Access: Breaking Host-GPU Bandwidth Bottlenecks in LLM Services","version":2},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-05-16T21:57:12.142867Z"},"links":{"citing_paper":"/paper/2512.16056"},"observation_digest":"sha256:c85b8f9466435198669a95f560bf0f3e906e11ea74bae42f00bc5a85cfca1c8d","observation_id":"9b6bc2ed-f353-49a8-8f2a-f66c4094a8f1","resolution":{"observed_at":"2026-05-16T21:58:36.521390Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Efficient intra-node hierarchical par- allelisms and dynamic load balancing strategies on het- erogeneous systems","venue":null,"work_id":"6bfe027c-74d3-46e8-950b-152a727420f4","year":2025},"citing_paper":{"arxiv_id":"2512.16056","last_updated":"2026-05-13T15:33:38Z","snapshot_observed_at":"2026-08-15T04:35:21.537340Z","submitted_at":"2025-12-18T00:45:00Z","title":"MultiPath Memory Access: Breaking Host-GPU Bandwidth Bottlenecks in LLM Services","version":2},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-05-16T21:57:12.142867Z"},"links":{"citing_paper":"/paper/2512.16056"},"observation_digest":"sha256:af7b8c2770fe9b8235d498b199f9c35de81551cc54bea6a7dfd57ea0103d9633","observation_id":"a7960dbc-5723-4890-9be3-85f620d00910","resolution":{"observed_at":"2026-05-16T21:58:36.515852Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Accelerating intra-node gpu communication: A perfor- mance model for multi-path transfers","venue":null,"work_id":"9210b26b-078a-47a2-85f3-86ab194fa000","year":2025},"citing_paper":{"arxiv_id":"2512.16056","last_updated":"2026-05-13T15:33:38Z","snapshot_observed_at":"2026-08-15T04:35:21.537340Z","submitted_at":"2025-12-18T00:45:00Z","title":"MultiPath Memory Access: Breaking Host-GPU Bandwidth Bottlenecks in LLM Services","version":2},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-05-16T21:57:12.142867Z"},"links":{"citing_paper":"/paper/2512.16056"},"observation_digest":"sha256:448acbe573ed278d54bfaab2d2c14d6b2f2d9648b21fc0de2aac60f0544b914d","observation_id":"b2c301ad-db6a-43f1-a76b-d1f37faf58f9","resolution":{"observed_at":"2026-05-16T21:58:36.524390Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Collabora- tive bandwidth-efficient intra-node allreduce","venue":null,"work_id":"2e1944a6-3d1a-4b67-bb97-322ce7767f2b","year":2025},"citing_paper":{"arxiv_id":"2512.16056","last_updated":"2026-05-13T15:33:38Z","snapshot_observed_at":"2026-08-15T04:35:21.537340Z","submitted_at":"2025-12-18T00:45:00Z","title":"MultiPath Memory Access: Breaking Host-GPU Bandwidth Bottlenecks in LLM Services","version":2},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-05-16T21:57:12.142867Z"},"links":{"citing_paper":"/paper/2512.16056"},"observation_digest":"sha256:1cc06f33747ef0d0d4f4bca11b3c6abc2afd5bdeda18acef6f7c7d8988c87e68","observation_id":"496af748-083d-4023-9184-ad22689cafbc","resolution":{"observed_at":"2026-05-16T21:58:36.510549Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Enhancing intra-node GPU-to-GPU perfor- mance in MPI+UCX through multi-path communication","venue":null,"work_id":"f249a011-9abc-4207-8a4c-269b26f37184","year":2024},"citing_paper":{"arxiv_id":"2512.16056","last_updated":"2026-05-13T15:33:38Z","snapshot_observed_at":"2026-08-15T04:35:21.537340Z","submitted_at":"2025-12-18T00:45:00Z","title":"MultiPath Memory Access: Breaking Host-GPU Bandwidth Bottlenecks in LLM Services","version":2},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-05-16T21:57:12.142867Z"},"links":{"citing_paper":"/paper/2512.16056"},"observation_digest":"sha256:20176ceec64c4952964027eb3f9f9575a80c7a7f5e528b42a1ffa4874442bf01","observation_id":"40fb733f-c921-4685-8dcc-06c691006112","resolution":{"observed_at":"2026-05-16T21:58:36.507739Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Sur- vey of intra-node gpu interconnection in scale-up net- work: Challenges, status, insights, and future directions","venue":null,"work_id":"30a8ef1c-03ef-425c-b94b-6f16d4609ab6","year":2025},"citing_paper":{"arxiv_id":"2512.16056","last_updated":"2026-05-13T15:33:38Z","snapshot_observed_at":"2026-08-15T04:35:21.537340Z","submitted_at":"2025-12-18T00:45:00Z","title":"MultiPath Memory Access: Breaking Host-GPU Bandwidth Bottlenecks in LLM Services","version":2},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-05-16T21:57:12.142867Z"},"links":{"citing_paper":"/paper/2512.16056"},"observation_digest":"sha256:79fee1b5737140f92ba98894cb091a3171e70fb3d5d43c07c8036a11a5611005","observation_id":"4615ee21-cc2f-4e5f-8846-f911100d65f1","resolution":{"observed_at":"2026-05-16T21:58:36.501381Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Engine-agnostic model hot-swapping for cost-effective llm inference","venue":null,"work_id":"5cb8bdf0-b222-4a06-83ed-5cc205e6137e","year":2025},"citing_paper":{"arxiv_id":"2512.16056","last_updated":"2026-05-13T15:33:38Z","snapshot_observed_at":"2026-08-15T04:35:21.537340Z","submitted_at":"2025-12-18T00:45:00Z","title":"MultiPath Memory Access: Breaking Host-GPU Bandwidth Bottlenecks in LLM Services","version":2},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-05-16T21:57:12.142867Z"},"links":{"citing_paper":"/paper/2512.16056"},"observation_digest":"sha256:9c9210fb9d6935bcead13e6c77b7e52648a52776408efef8159831ce1dbd53b8","observation_id":"7d2bb17b-3976-4b20-bba6-05076a2c7257","resolution":{"observed_at":"2026-05-16T21:58:36.504565Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2511.04677","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-30T18:14:59.286239Z","title":"Scalable and efficient intra-and inter-node in- terconnection networks for post-exascale supercomput- ers and data centers","venue":null,"work_id":"07a89d25-1be2-41a9-bbfc-8a23b4e860f7","year":2025},"citing_paper":{"arxiv_id":"2512.16056","last_updated":"2026-05-13T15:33:38Z","snapshot_observed_at":"2026-08-15T04:35:21.537340Z","submitted_at":"2025-12-18T00:45:00Z","title":"MultiPath Memory Access: Breaking Host-GPU Bandwidth Bottlenecks in LLM Services","version":2},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-05-16T21:57:12.142867Z"},"links":{"citing_paper":"/paper/2512.16056"},"observation_digest":"sha256:7f0285eefbad37ac462c96273267538769cc42a37938f45657928038f5bd8b3a","observation_id":"da24bbc2-1653-4d9b-a3bf-4962e6087438","resolution":{"observed_at":"2026-05-16T21:58:35.740286Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2502.20965","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Under- standing intra-node communication in hpc systems and datacenters","venue":null,"work_id":"a3b76b16-b786-4e65-a534-22fed7eca7f6","year":2025},"citing_paper":{"arxiv_id":"2512.16056","last_updated":"2026-05-13T15:33:38Z","snapshot_observed_at":"2026-08-15T04:35:21.537340Z","submitted_at":"2025-12-18T00:45:00Z","title":"MultiPath Memory Access: Breaking Host-GPU Bandwidth Bottlenecks in LLM Services","version":2},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-05-16T21:57:12.142867Z"},"links":{"citing_paper":"/paper/2512.16056"},"observation_digest":"sha256:a82d508cd1817c4d614c8a6b9b7917088e8151c8b6e6fe73a8f3485985e61012","observation_id":"623f91de-72ee-4c44-98d6-e34d6b100ede","resolution":{"observed_at":"2026-05-16T21:58:35.733063Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Efficient multi-path NVLink/PCIe-aware UCX-based collective communi- cation for deep learning","venue":null,"work_id":"b1160dec-eafb-4291-9660-7a24bf6ae98a","year":2021},"citing_paper":{"arxiv_id":"2512.16056","last_updated":"2026-05-13T15:33:38Z","snapshot_observed_at":"2026-08-15T04:35:21.537340Z","submitted_at":"2025-12-18T00:45:00Z","title":"MultiPath Memory Access: Breaking Host-GPU Bandwidth Bottlenecks in LLM Services","version":2},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-05-16T21:57:12.142867Z"},"links":{"citing_paper":"/paper/2512.16056"},"observation_digest":"sha256:d7498c0e4d656c2fb628fafe75042ada855fc021420adc3767812bd74b185b92","observation_id":"1f8a2a01-f2bf-4fbc-bed3-14fa25c98d45","resolution":{"observed_at":"2026-05-16T21:58:36.513247Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Performance models for cpu-gpu data transfers","venue":null,"work_id":"b2c0ad2c-7a5d-4441-990d-40eae11adc34","year":2014},"citing_paper":{"arxiv_id":"2512.16056","last_updated":"2026-05-13T15:33:38Z","snapshot_observed_at":"2026-08-15T04:35:21.537340Z","submitted_at":"2025-12-18T00:45:00Z","title":"MultiPath Memory Access: Breaking Host-GPU Bandwidth Bottlenecks in LLM Services","version":2},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-05-16T21:57:12.142867Z"},"links":{"citing_paper":"/paper/2512.16056"},"observation_digest":"sha256:bc23f8def7328616cfc34e38224ca20bd8edf9bf678249f9a0870ea4b989ad38","observation_id":"8102c47c-503d-49b4-945d-10e1e3a2f206","resolution":{"observed_at":"2026-05-16T21:58:36.518719Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Sleep Mode","venue":null,"work_id":"089b01c6-05e2-4bc0-8c56-279fa8698639","year":null},"citing_paper":{"arxiv_id":"2512.16056","last_updated":"2026-05-13T15:33:38Z","snapshot_observed_at":"2026-08-15T04:35:21.537340Z","submitted_at":"2025-12-18T00:45:00Z","title":"MultiPath Memory Access: Breaking Host-GPU Bandwidth Bottlenecks in LLM Services","version":2},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-05-16T21:57:12.142867Z"},"links":{"citing_paper":"/paper/2512.16056"},"observation_digest":"sha256:6421743bc2af3e1e7a19bdb094d6dadb8979ae7f8b72738b0076a65bc7dd4bdd","observation_id":"14b70501-5ab6-42fa-86ee-ac22a3efbf95","resolution":{"observed_at":"2026-05-16T21:58:36.526863Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Design, implementation and evalua- tion of congestion control for multipath {TCP}","venue":null,"work_id":"94256e85-85de-492f-8f05-211e5ca17a43","year":2011},"citing_paper":{"arxiv_id":"2512.16056","last_updated":"2026-05-13T15:33:38Z","snapshot_observed_at":"2026-08-15T04:35:21.537340Z","submitted_at":"2025-12-18T00:45:00Z","title":"MultiPath Memory Access: Breaking Host-GPU Bandwidth Bottlenecks in LLM Services","version":2},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-05-16T21:57:12.142867Z"},"links":{"citing_paper":"/paper/2512.16056"},"observation_digest":"sha256:10a06767ca253f38e5ac7360da6348dfd481e25d0db686d6296ff6ec991ec028","observation_id":"cdaa05ba-84ab-49e9-be66-8a0e38219325","resolution":{"observed_at":"2026-05-16T21:58:36.576292Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.09388","last_updated":"2025-05-14T13:41:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-05-14T13:41:34Z","title":"Qwen3 Technical Report","version":1},"cited_work":{"arxiv_id":"2505.09388","doi":"10.1016/j.aiopen.2022.12","metadata_source":"pith","pith_arxiv_id":"2505.09388","snapshot_observed_at":"2026-07-11T11:50:26.030339Z","title":"Qwen3 Technical Report","venue":"cs.CL","work_id":"25a4e30c-1232-48e7-9925-02fa12ba7c9e","year":2025},"citing_paper":{"arxiv_id":"2512.16056","last_updated":"2026-05-13T15:33:38Z","snapshot_observed_at":"2026-08-15T04:35:21.537340Z","submitted_at":"2025-12-18T00:45:00Z","title":"MultiPath Memory Access: Breaking Host-GPU Bandwidth Bottlenecks in LLM Services","version":2},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-05-16T21:57:12.142867Z"},"links":{"cited_paper":"/paper/2505.09388","citing_paper":"/paper/2512.16056"},"observation_digest":"sha256:c94198a6f77f7ec704d5d374936770f37ae730f51a918cbed20e4523dd2f30f1","observation_id":"7e193dbb-1e27-4f50-bbaf-379040edb276","resolution":{"observed_at":"2026-05-16T21:58:35.751577Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Learned prefix caching for efficient llm inference","venue":null,"work_id":"2b0020c3-5fd6-464b-a1fa-de83e0d84c52","year":null},"citing_paper":{"arxiv_id":"2512.16056","last_updated":"2026-05-13T15:33:38Z","snapshot_observed_at":"2026-08-15T04:35:21.537340Z","submitted_at":"2025-12-18T00:45:00Z","title":"MultiPath Memory Access: Breaking Host-GPU Bandwidth Bottlenecks in LLM Services","version":2},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-05-16T21:57:12.142867Z"},"links":{"citing_paper":"/paper/2512.16056"},"observation_digest":"sha256:e11a747f695e12a84c24f3a59e256a4ead8f2c141be8383bd11a33e3a695ad0a","observation_id":"abca3b2d-459c-4ccb-8696-d3852c16c6df","resolution":{"observed_at":"2026-05-16T21:58:36.542420Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Sglang: Efficient execution of structured language model programs.Advances in neural information pro- cessing systems, 37:62557–62583","venue":null,"work_id":"c681120c-434f-4322-ba9b-3a920da9a3f1","year":2024},"citing_paper":{"arxiv_id":"2512.16056","last_updated":"2026-05-13T15:33:38Z","snapshot_observed_at":"2026-08-15T04:35:21.537340Z","submitted_at":"2025-12-18T00:45:00Z","title":"MultiPath Memory Access: Breaking Host-GPU Bandwidth Bottlenecks in LLM Services","version":2},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-05-16T21:57:12.142867Z"},"links":{"citing_paper":"/paper/2512.16056"},"observation_digest":"sha256:2245f32d2540f8b29398110dc9b461d49b72754946da9fa7c09341def9a1b46f","observation_id":"272a3b40-91b7-49b6-b0ad-88847301d1ef","resolution":{"observed_at":"2026-05-16T21:58:36.544891Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"{DistServe}: Disaggregating prefill and decoding for goodput-optimized large language model serving","venue":null,"work_id":"e525a031-2049-4618-9cf7-54bd6aaf9178","year":2024},"citing_paper":{"arxiv_id":"2512.16056","last_updated":"2026-05-13T15:33:38Z","snapshot_observed_at":"2026-08-15T04:35:21.537340Z","submitted_at":"2025-12-18T00:45:00Z","title":"MultiPath Memory Access: Breaking Host-GPU Bandwidth Bottlenecks in LLM Services","version":2},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-05-16T21:57:12.142867Z"},"links":{"citing_paper":"/paper/2512.16056"},"observation_digest":"sha256:dfe23ade0e266113a43a315138767e82f38fec1d514aaddb6623d8df82c9913f","observation_id":"77adca2e-93ff-407f-97cf-6c52547b147b","resolution":{"observed_at":"2026-05-16T21:58:36.498710Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2512.16056","last_updated":"2026-05-13T15:33:38Z","latest_version":2,"primary_category":"cs.DC","snapshot_observed_at":"2026-08-15T04:35:21.537340Z","submitted_at":"2025-12-18T00:45:00Z","title":"MultiPath Memory Access: Breaking Host-GPU Bandwidth Bottlenecks in LLM Services"},"reference_resolution":{"displayed":52,"state_counts":{"malformed_identifier":0,"metadata_mismatch":1,"parse_uncertain":0,"unresolved":1,"verified_exact":9,"verified_fuzzy":41},"total_outbound_references":52},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"thesis":"As of 18 August 2026, this Paper Citation Record lists 52 of 52 outbound references and 1 inbound Pith citation observation for arXiv:2512.16056."}