{"as_of":"2026-08-06T03:29:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:3dc009d36e49dfb1ba453e3e2041c53c02712b9a9bff77ed34209046972fa2a2","coverage":[{"denominator":60,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":60,"source":"paper_references, paper_reference_links","source_observed_at":"2026-07-13T22:20:51.838309Z","state":"measured"},{"denominator":71,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":71,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-05T06:32:48.257954+00:00","state":"measured"},{"denominator":11,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":11,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-02T13:00:11.112294Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":1,"source":"pith","source_observed_at":"2026-08-05T02:28:24.338817Z","state":"measured"}],"external_citation_measurements":[{"count":0,"observed_at":"2026-08-05T02:28:24.338817Z","source":"pith"}],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2603.18897","last_updated":"2026-06-16T09:34:16Z","snapshot_observed_at":"2026-07-13T22:20:51.137198Z","submitted_at":"2026-03-19T13:36:50Z","title":"Parallelizing Tool Execution and LLM Generation for Low-Latency Agent Serving","version":3},"cited_work":{"arxiv_id":"2603.18897","doi":"10.48550/arxiv.2603.18897","metadata_source":"pith","pith_arxiv_id":"2603.18897","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Act while thinking: Accelerating llm agents via pattern-aware speculative tool execution","venue":"cs.DC","work_id":"3562f15c-5f03-4fcb-a0c4-018fd064a4d1","year":2026},"citing_paper":{"arxiv_id":"2604.08224","last_updated":"2026-04-09T13:19:41Z","snapshot_observed_at":"2026-08-02T22:56:00.067549Z","submitted_at":"2026-04-09T13:19:41Z","title":"Externalization in LLM Agents: A Unified Review of Memory, Skills, Protocols and Harness Engineering","version":1},"reference_index":132,"source":"arxiv_source","source_observed_at":"2026-05-10T17:40:14.733882Z"},"links":{"cited_paper":"/paper/2603.18897","citing_paper":"/paper/2604.08224"},"observation_digest":"sha256:6108b2e45c4f2db66644efcce977bc72e989e4257fd5616791220bc4a09ea222","observation_id":"2dfef6be-35bd-42a0-9a4f-690beff956d8","resolution":{"observed_at":"2026-06-19T17:09:53.145794Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2603.18897","last_updated":"2026-06-16T09:34:16Z","snapshot_observed_at":"2026-07-13T22:20:51.137198Z","submitted_at":"2026-03-19T13:36:50Z","title":"Parallelizing Tool Execution and LLM Generation for Low-Latency Agent Serving","version":3},"cited_work":{"arxiv_id":"2603.18897","doi":"10.48550/arxiv.2603.18897","metadata_source":"pith","pith_arxiv_id":"2603.18897","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Act while thinking: Accelerating llm agents via pattern-aware speculative tool execution","venue":"cs.DC","work_id":"3562f15c-5f03-4fcb-a0c4-018fd064a4d1","year":2026},"citing_paper":{"arxiv_id":"2604.14228","last_updated":"2026-07-02T17:12:13Z","snapshot_observed_at":"2026-07-12T21:00:43.195941Z","submitted_at":"2026-04-14T17:59:37Z","title":"Dive into Claude Code: The Design Space of Today's and Future AI Agent Systems","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-05-10T14:30:30.311285Z"},"links":{"cited_paper":"/paper/2603.18897","citing_paper":"/paper/2604.14228"},"observation_digest":"sha256:30c11f31195a36da27da1f6f9e11eb658daafcabc1a2df4fe62cc289158fbac3","observation_id":"54bca6fb-4282-49e6-900a-380d5537ded3","resolution":{"observed_at":"2026-06-19T17:09:53.145794Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2603.18897","last_updated":"2026-06-16T09:34:16Z","snapshot_observed_at":"2026-07-13T22:20:51.137198Z","submitted_at":"2026-03-19T13:36:50Z","title":"Parallelizing Tool Execution and LLM Generation for Low-Latency Agent Serving","version":3},"cited_work":{"arxiv_id":"2603.18897","doi":"10.48550/arxiv.2603.18897","metadata_source":"pith","pith_arxiv_id":"2603.18897","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Act while thinking: Accelerating llm agents via pattern-aware speculative tool execution","venue":"cs.DC","work_id":"3562f15c-5f03-4fcb-a0c4-018fd064a4d1","year":2026},"citing_paper":{"arxiv_id":"2604.16469","last_updated":"2026-04-09T07:42:17Z","snapshot_observed_at":"2026-08-03T10:17:06.378696Z","submitted_at":"2026-04-09T07:42:17Z","title":"B-PASTE: Beam-Aware Pattern-Guided Speculative Execution for Resource-Constrained LLM Agents","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-10T17:12:56.274633Z"},"links":{"cited_paper":"/paper/2603.18897","citing_paper":"/paper/2604.16469"},"observation_digest":"sha256:c86f85a2a8bddb711a2e0fb72b9bb7a90c76c39710aa3fe5371c2f794cef258c","observation_id":"6680dad4-98db-4de8-a70f-1b910d01f3a3","resolution":{"observed_at":"2026-06-19T17:09:53.145794Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2603.18897","last_updated":"2026-06-16T09:34:16Z","snapshot_observed_at":"2026-07-13T22:20:51.137198Z","submitted_at":"2026-03-19T13:36:50Z","title":"Parallelizing Tool Execution and LLM Generation for Low-Latency Agent Serving","version":3},"cited_work":{"arxiv_id":"2603.18897","doi":"10.48550/arxiv.2603.18897","metadata_source":"pith","pith_arxiv_id":"2603.18897","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Act while thinking: Accelerating llm agents via pattern-aware speculative tool execution","venue":"cs.DC","work_id":"3562f15c-5f03-4fcb-a0c4-018fd064a4d1","year":2026},"citing_paper":{"arxiv_id":"2604.20938","last_updated":"2026-04-22T13:45:12Z","snapshot_observed_at":"2026-07-06T23:07:36.996629Z","submitted_at":"2026-04-22T13:45:12Z","title":"HARBOR: Automated Harness Optimization","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-05-10T00:39:49.572962Z"},"links":{"cited_paper":"/paper/2603.18897","citing_paper":"/paper/2604.20938"},"observation_digest":"sha256:648d1cc5f0a683d1f5cf97720f8b3180ab675feda3f9a21e8651c329af670497","observation_id":"292f0524-d218-4e62-9b83-5efb49c29312","resolution":{"observed_at":"2026-06-19T17:09:53.145794Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2603.18897","last_updated":"2026-06-16T09:34:16Z","snapshot_observed_at":"2026-07-13T22:20:51.137198Z","submitted_at":"2026-03-19T13:36:50Z","title":"Parallelizing Tool Execution and LLM Generation for Low-Latency Agent Serving","version":3},"cited_work":{"arxiv_id":"2603.18897","doi":"10.48550/arxiv.2603.18897","metadata_source":"pith","pith_arxiv_id":"2603.18897","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Act while thinking: Accelerating llm agents via pattern-aware speculative tool execution","venue":"cs.DC","work_id":"3562f15c-5f03-4fcb-a0c4-018fd064a4d1","year":2026},"citing_paper":{"arxiv_id":"2605.06472","last_updated":"2026-05-07T15:57:51Z","snapshot_observed_at":"2026-08-04T11:44:04.290481Z","submitted_at":"2026-05-07T15:57:51Z","title":"Efficient Serving for Dynamic Agent Workflows with Prediction-based KV-Cache Management","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-05-08T12:40:33.844220Z"},"links":{"cited_paper":"/paper/2603.18897","citing_paper":"/paper/2605.06472"},"observation_digest":"sha256:28c08e31b2f5a6d2e50710c086b7acbb7b311ee861be3bb1340dddf07e609f91","observation_id":"55e89ddf-6291-4c94-a866-5e91a980a5ba","resolution":{"observed_at":"2026-06-19T17:09:53.145794Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2603.18897","last_updated":"2026-06-16T09:34:16Z","snapshot_observed_at":"2026-07-13T22:20:51.137198Z","submitted_at":"2026-03-19T13:36:50Z","title":"Parallelizing Tool Execution and LLM Generation for Low-Latency Agent Serving","version":3},"cited_work":{"arxiv_id":"2603.18897","doi":"10.48550/arxiv.2603.18897","metadata_source":"pith","pith_arxiv_id":"2603.18897","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Act while thinking: Accelerating llm agents via pattern-aware speculative tool execution","venue":"cs.DC","work_id":"3562f15c-5f03-4fcb-a0c4-018fd064a4d1","year":2026},"citing_paper":{"arxiv_id":"2605.21965","last_updated":"2026-05-21T03:55:47Z","snapshot_observed_at":"2026-07-06T23:32:20.708664Z","submitted_at":"2026-05-21T03:55:47Z","title":"SpecHop: Continuous Speculation for Accelerating Multi-Hop Retrieval Agents","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-05-22T06:50:06.933671Z"},"links":{"cited_paper":"/paper/2603.18897","citing_paper":"/paper/2605.21965"},"observation_digest":"sha256:35a2132bf96ed382c19d9d7f67b55021625f0a0e30fee94222b7cb63befd7f86","observation_id":"7ef5f191-a46d-4c32-92c4-f324b2684b88","resolution":{"observed_at":"2026-06-19T17:09:53.145794Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2603.18897","last_updated":"2026-06-16T09:34:16Z","snapshot_observed_at":"2026-07-13T22:20:51.137198Z","submitted_at":"2026-03-19T13:36:50Z","title":"Parallelizing Tool Execution and LLM Generation for Low-Latency Agent Serving","version":3},"cited_work":{"arxiv_id":"2603.18897","doi":"10.48550/arxiv.2603.18897","metadata_source":"pith","pith_arxiv_id":"2603.18897","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Act while thinking: Accelerating llm agents via pattern-aware speculative tool execution","venue":"cs.DC","work_id":"3562f15c-5f03-4fcb-a0c4-018fd064a4d1","year":2026},"citing_paper":{"arxiv_id":"2605.27744","last_updated":"2026-07-30T01:02:31Z","snapshot_observed_at":"2026-08-05T15:47:59.460314Z","submitted_at":"2026-05-26T22:38:34Z","title":"A Policy-Driven Runtime Layer for Agentic LLM Serving","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-06-29T16:54:36.387432Z"},"links":{"cited_paper":"/paper/2603.18897","citing_paper":"/paper/2605.27744"},"observation_digest":"sha256:5fcf2b31fe8b84e1e82c4faa69711e00ffe93d655a8b977b874b18ca87583584","observation_id":"0847f434-3a1b-4d45-845c-1f986df5c004","resolution":{"observed_at":"2026-06-29T17:03:41.401601Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2603.18897","last_updated":"2026-06-16T09:34:16Z","snapshot_observed_at":"2026-07-13T22:20:51.137198Z","submitted_at":"2026-03-19T13:36:50Z","title":"Parallelizing Tool Execution and LLM Generation for Low-Latency Agent Serving","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2603.18897","snapshot_observed_at":"2026-08-02T13:00:11.112294Z","title":null,"venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2605.27744","last_updated":"2026-07-30T01:02:31Z","snapshot_observed_at":"2026-08-05T15:47:59.460314Z","submitted_at":"2026-05-26T22:38:34Z","title":"A Policy-Driven Runtime Layer for Agentic LLM Serving","version":2},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-02T13:00:11.112294Z"},"links":{"cited_paper":"/paper/2603.18897","citing_paper":"/paper/2605.27744"},"observation_digest":"sha256:c6acc6340717f74855061c27841c5905d9de25a88e49be9807b60be5cf57e454","observation_id":"58d0c711-612a-4e93-a9b0-ed8f2155283e","resolution":{"observed_at":"2026-08-02T13:00:11.112294Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2603.18897","last_updated":"2026-06-16T09:34:16Z","snapshot_observed_at":"2026-07-13T22:20:51.137198Z","submitted_at":"2026-03-19T13:36:50Z","title":"Parallelizing Tool Execution and LLM Generation for Low-Latency Agent Serving","version":3},"cited_work":{"arxiv_id":"2603.18897","doi":"10.48550/arxiv.2603.18897","metadata_source":"pith","pith_arxiv_id":"2603.18897","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Act while thinking: Accelerating llm agents via pattern-aware speculative tool execution","venue":"cs.DC","work_id":"3562f15c-5f03-4fcb-a0c4-018fd064a4d1","year":2026},"citing_paper":{"arxiv_id":"2606.00866","last_updated":"2026-05-30T19:44:25Z","snapshot_observed_at":"2026-08-02T13:05:04.826857Z","submitted_at":"2026-05-30T19:44:25Z","title":"Idleness is Relative: Exploiting Tool-Call Idle Windows for Offloading in Agentic Systems with MORI","version":1},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-06-28T17:30:56.324289Z"},"links":{"cited_paper":"/paper/2603.18897","citing_paper":"/paper/2606.00866"},"observation_digest":"sha256:51f8a0816f5f12ea4f51427bfb1c9ae31bb24df2448717c616d0cd93ee2a24af","observation_id":"32e147c3-e979-442c-96ce-25bf1db92902","resolution":{"observed_at":"2026-07-01T21:06:13.596214Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2603.18897","last_updated":"2026-06-16T09:34:16Z","snapshot_observed_at":"2026-07-13T22:20:51.137198Z","submitted_at":"2026-03-19T13:36:50Z","title":"Parallelizing Tool Execution and LLM Generation for Low-Latency Agent Serving","version":3},"cited_work":{"arxiv_id":"2603.18897","doi":"10.48550/arxiv.2603.18897","metadata_source":"pith","pith_arxiv_id":"2603.18897","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Act while thinking: Accelerating llm agents via pattern-aware speculative tool execution","venue":"cs.DC","work_id":"3562f15c-5f03-4fcb-a0c4-018fd064a4d1","year":2026},"citing_paper":{"arxiv_id":"2606.02483","last_updated":"2026-06-01T16:53:19Z","snapshot_observed_at":"2026-07-06T23:42:53.514946Z","submitted_at":"2026-06-01T16:53:19Z","title":"Ghost Tool Calls: Issue-Time Privacy for Speculative Agent Tools","version":1},"reference_index":2,"source":"arxiv_source","source_observed_at":"2026-06-28T13:46:11.765116Z"},"links":{"cited_paper":"/paper/2603.18897","citing_paper":"/paper/2606.02483"},"observation_digest":"sha256:f382f497d84f5019a12137104f18e672717353aff8d371e303c83a9aeb51583b","observation_id":"4fea55bb-f4d3-4485-8bae-60919055f378","resolution":{"observed_at":"2026-06-28T14:42:18.259022Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2603.18897","last_updated":"2026-06-16T09:34:16Z","snapshot_observed_at":"2026-07-13T22:20:51.137198Z","submitted_at":"2026-03-19T13:36:50Z","title":"Parallelizing Tool Execution and LLM Generation for Low-Latency Agent Serving","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2603.18897","snapshot_observed_at":"2026-07-12T03:14:02.219678Z","title":null,"venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2607.03333","last_updated":"2026-07-03T13:51:32Z","snapshot_observed_at":"2026-08-04T06:12:58.772470Z","submitted_at":"2026-07-03T13:51:32Z","title":"SPORK: Self-Speculative Forking to Accelerate Agentic LLM Inference","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-07-12T03:14:02.219678Z"},"links":{"cited_paper":"/paper/2603.18897","citing_paper":"/paper/2607.03333"},"observation_digest":"sha256:80d58fee71be3f29f9c2c5dd12a249ad8ce3390559bd5763b9faaae299a16bbf","observation_id":"59050678-5f20-4fea-a87c-04fdc1f525a4","resolution":{"observed_at":"2026-07-12T03:14:02.219678Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2603.18897/citation-record","integrity":"/paper/2603.18897/integrity","json":"/paper/2603.18897/citation-record.json","paper":"/paper/2603.18897"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-07-13T22:20:51.838309Z","title":"Agent Skills","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2603.18897","last_updated":"2026-06-16T09:34:16Z","snapshot_observed_at":"2026-07-13T22:20:51.137198Z","submitted_at":"2026-03-19T13:36:50Z","title":"Parallelizing Tool Execution and LLM Generation for Low-Latency Agent Serving","version":3},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-07-13T22:20:51.838309Z"},"links":{"citing_paper":"/paper/2603.18897"},"observation_digest":"sha256:c5782a2268349b680fa6fed782f801271748e110c3f96416906c712f46c6395d","observation_id":"3f2b157a-9292-4ecf-a0be-3d4280bf016e","resolution":{"observed_at":"2026-07-13T22:20:51.838309Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-07-13T22:20:51.838309Z","title":"Build, Debug & Deploy with AI","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2603.18897","last_updated":"2026-06-16T09:34:16Z","snapshot_observed_at":"2026-07-13T22:20:51.137198Z","submitted_at":"2026-03-19T13:36:50Z","title":"Parallelizing Tool Execution and LLM Generation for Low-Latency Agent Serving","version":3},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-07-13T22:20:51.838309Z"},"links":{"citing_paper":"/paper/2603.18897"},"observation_digest":"sha256:902d18d16944ddea37fa7d624f797d559eddb30672d58d33c5d5e4f73286963c","observation_id":"e429b776-38d4-4d7e-b830-93417480c39f","resolution":{"observed_at":"2026-07-13T22:20:51.838309Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-07-13T22:20:51.838309Z","title":"Claude Code | Claude","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2603.18897","last_updated":"2026-06-16T09:34:16Z","snapshot_observed_at":"2026-07-13T22:20:51.137198Z","submitted_at":"2026-03-19T13:36:50Z","title":"Parallelizing Tool Execution and LLM Generation for Low-Latency Agent Serving","version":3},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-07-13T22:20:51.838309Z"},"links":{"citing_paper":"/paper/2603.18897"},"observation_digest":"sha256:0ed769b4d626c6b10734d66349dfb82920f9c56d3f49bb26a9263e02eedf0f4d","observation_id":"44088e15-ace4-4967-a85b-737020a0c975","resolution":{"observed_at":"2026-07-13T22:20:51.838309Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-07-13T22:20:51.838309Z","title":"GitHub Copilot","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2603.18897","last_updated":"2026-06-16T09:34:16Z","snapshot_observed_at":"2026-07-13T22:20:51.137198Z","submitted_at":"2026-03-19T13:36:50Z","title":"Parallelizing Tool Execution and LLM Generation for Low-Latency Agent Serving","version":3},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-07-13T22:20:51.838309Z"},"links":{"citing_paper":"/paper/2603.18897"},"observation_digest":"sha256:dd4f420a3a1f664b2d9a28f73fac0955b9d1a815183c4812a8d55aca59be6333","observation_id":"9fa628e8-e646-45d3-92fe-7f879a1f69db","resolution":{"observed_at":"2026-07-13T22:20:51.838309Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-07-13T22:20:51.838309Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2603.18897","last_updated":"2026-06-16T09:34:16Z","snapshot_observed_at":"2026-07-13T22:20:51.137198Z","submitted_at":"2026-03-19T13:36:50Z","title":"Parallelizing Tool Execution and LLM Generation for Low-Latency Agent Serving","version":3},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-07-13T22:20:51.838309Z"},"links":{"citing_paper":"/paper/2603.18897"},"observation_digest":"sha256:5ef52eeddb2402fc00fa99f21f5e987959ee8f39188301d1b9773f487e775efc","observation_id":"f96b1df7-ddb4-4fb8-b349-5812896c3aa0","resolution":{"observed_at":"2026-07-13T22:20:51.838309Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-07-13T22:20:51.838309Z","title":"InProceedings of the 41st International Conference on Machine Learning (Proceedings of Machine Learning Research, Vol","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2603.18897","last_updated":"2026-06-16T09:34:16Z","snapshot_observed_at":"2026-07-13T22:20:51.137198Z","submitted_at":"2026-03-19T13:36:50Z","title":"Parallelizing Tool Execution and LLM Generation for Low-Latency Agent Serving","version":3},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-07-13T22:20:51.838309Z"},"links":{"citing_paper":"/paper/2603.18897"},"observation_digest":"sha256:0c7f7015489054103a15fc87a60fea2f29f53d3f0b203e9c3759f339cc584ede","observation_id":"5c23a44f-8640-4add-b83d-9d65ce2dce44","resolution":{"observed_at":"2026-07-13T22:20:51.838309Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-07-13T22:20:51.838309Z","title":"Gulavani, Alexey Tumanov, and Ramachandran Ramjee","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2603.18897","last_updated":"2026-06-16T09:34:16Z","snapshot_observed_at":"2026-07-13T22:20:51.137198Z","submitted_at":"2026-03-19T13:36:50Z","title":"Parallelizing Tool Execution and LLM Generation for Low-Latency Agent Serving","version":3},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-07-13T22:20:51.838309Z"},"links":{"citing_paper":"/paper/2603.18897"},"observation_digest":"sha256:55e867ec930d3d51b29bc18f7ac6c23f0641ceb5925c3a5374774d35c0fb5c02","observation_id":"1b298f25-0119-487e-a5cc-6e9ab16316cf","resolution":{"observed_at":"2026-07-13T22:20:51.838309Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-07-13T22:20:51.838309Z","title":"2025.DeepResearch: Tongyi Deep Research, the Leading Open-source Deep Research Agent","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2603.18897","last_updated":"2026-06-16T09:34:16Z","snapshot_observed_at":"2026-07-13T22:20:51.137198Z","submitted_at":"2026-03-19T13:36:50Z","title":"Parallelizing Tool Execution and LLM Generation for Low-Latency Agent Serving","version":3},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-07-13T22:20:51.838309Z"},"links":{"citing_paper":"/paper/2603.18897"},"observation_digest":"sha256:4cd76f9ee61c438a901346524a91a1c2508d166637b002fd44717e3ee2fbd68c","observation_id":"7d37ff13-1672-4a9d-9343-89312760dbb0","resolution":{"observed_at":"2026-07-13T22:20:51.838309Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.14199","last_updated":"2024-11-21T15:07:42Z","snapshot_observed_at":"2026-07-06T19:53:47.605701Z","submitted_at":"2024-11-21T15:07:42Z","title":"OpenScholar: Synthesizing Scientific Literature with Retrieval-augmented LMs","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.14199","snapshot_observed_at":"2026-07-13T22:20:51.838309Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2603.18897","last_updated":"2026-06-16T09:34:16Z","snapshot_observed_at":"2026-07-13T22:20:51.137198Z","submitted_at":"2026-03-19T13:36:50Z","title":"Parallelizing Tool Execution and LLM Generation for Low-Latency Agent Serving","version":3},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-07-13T22:20:51.838309Z"},"links":{"cited_paper":"/paper/2411.14199","citing_paper":"/paper/2603.18897"},"observation_digest":"sha256:e57d869d6d94e772b801c8490b2dda98b529c635cc5156ad88595823bf5eb0c2","observation_id":"d84c5835-c6e9-4150-8e41-72c195624b01","resolution":{"observed_at":"2026-07-13T22:20:51.838309Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-07-13T22:20:51.838309Z","title":"Lee, Deming Chen, and Tri Dao","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2603.18897","last_updated":"2026-06-16T09:34:16Z","snapshot_observed_at":"2026-07-13T22:20:51.137198Z","submitted_at":"2026-03-19T13:36:50Z","title":"Parallelizing Tool Execution and LLM Generation for Low-Latency Agent Serving","version":3},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-07-13T22:20:51.838309Z"},"links":{"citing_paper":"/paper/2603.18897"},"observation_digest":"sha256:9f563dd4e9d4ba77de3ee735e7c08c888352bc48763d3e6569723faf19241c44","observation_id":"5e8f79ec-8278-45de-98e9-ba3bb09ecd38","resolution":{"observed_at":"2026-07-13T22:20:51.838309Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-07-13T22:20:51.838309Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2603.18897","last_updated":"2026-06-16T09:34:16Z","snapshot_observed_at":"2026-07-13T22:20:51.137198Z","submitted_at":"2026-03-19T13:36:50Z","title":"Parallelizing Tool Execution and LLM Generation for Low-Latency Agent Serving","version":3},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-07-13T22:20:51.838309Z"},"links":{"citing_paper":"/paper/2603.18897"},"observation_digest":"sha256:35038972eef6b4a5b41c877495980d58ef6b9f9a19c9ac92c638a28b655405ca","observation_id":"e6fd1a64-beec-41e5-bf9f-a72144609b64","resolution":{"observed_at":"2026-07-13T22:20:51.838309Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-07-13T22:20:51.838309Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2603.18897","last_updated":"2026-06-16T09:34:16Z","snapshot_observed_at":"2026-07-13T22:20:51.137198Z","submitted_at":"2026-03-19T13:36:50Z","title":"Parallelizing Tool Execution and LLM Generation for Low-Latency Agent Serving","version":3},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-07-13T22:20:51.838309Z"},"links":{"citing_paper":"/paper/2603.18897"},"observation_digest":"sha256:3f4959c682ad2d860ae8f9047aa0d2df864982bb49dd10d754c7de67b4bc51a1","observation_id":"999c972a-2157-41cd-8248-6798bcb51f34","resolution":{"observed_at":"2026-07-13T22:20:51.838309Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-07-13T22:20:51.838309Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2603.18897","last_updated":"2026-06-16T09:34:16Z","snapshot_observed_at":"2026-07-13T22:20:51.137198Z","submitted_at":"2026-03-19T13:36:50Z","title":"Parallelizing Tool Execution and LLM Generation for Low-Latency Agent Serving","version":3},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-07-13T22:20:51.838309Z"},"links":{"citing_paper":"/paper/2603.18897"},"observation_digest":"sha256:dfa91f70c9bf55cd84162a4933baf953c224bd288fdac5a8d0e511649594c1c9","observation_id":"90192f59-3fa5-44a5-81cb-f3fa8c40be42","resolution":{"observed_at":"2026-07-13T22:20:51.838309Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-07-13T22:20:51.838309Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2603.18897","last_updated":"2026-06-16T09:34:16Z","snapshot_observed_at":"2026-07-13T22:20:51.137198Z","submitted_at":"2026-03-19T13:36:50Z","title":"Parallelizing Tool Execution and LLM Generation for Low-Latency Agent Serving","version":3},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-07-13T22:20:51.838309Z"},"links":{"citing_paper":"/paper/2603.18897"},"observation_digest":"sha256:176872d649b7dcf61047d1b9c49a7ac5e9d0689466304829288f1d33233fe838","observation_id":"9ab94479-7c1a-433c-bccd-aafed1b40541","resolution":{"observed_at":"2026-07-13T22:20:51.838309Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-07-13T22:20:51.838309Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2603.18897","last_updated":"2026-06-16T09:34:16Z","snapshot_observed_at":"2026-07-13T22:20:51.137198Z","submitted_at":"2026-03-19T13:36:50Z","title":"Parallelizing Tool Execution and LLM Generation for Low-Latency Agent Serving","version":3},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-07-13T22:20:51.838309Z"},"links":{"citing_paper":"/paper/2603.18897"},"observation_digest":"sha256:62799899b080121dcdde717143160c6138f581dcee76a4612432479a8ebe8c5a","observation_id":"a88f30ba-ffde-4a43-a842-7400346ec11d","resolution":{"observed_at":"2026-07-13T22:20:51.838309Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-07-13T22:20:51.838309Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2603.18897","last_updated":"2026-06-16T09:34:16Z","snapshot_observed_at":"2026-07-13T22:20:51.137198Z","submitted_at":"2026-03-19T13:36:50Z","title":"Parallelizing Tool Execution and LLM Generation for Low-Latency Agent Serving","version":3},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-07-13T22:20:51.838309Z"},"links":{"citing_paper":"/paper/2603.18897"},"observation_digest":"sha256:c8eac2b018c15d76aa3b9a32307b7f59474644f1112ecfe7b3a97648f8172dd1","observation_id":"e04455e2-de4c-4f73-9b53-80dd568a5365","resolution":{"observed_at":"2026-07-13T22:20:51.838309Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-07-13T22:20:51.838309Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2603.18897","last_updated":"2026-06-16T09:34:16Z","snapshot_observed_at":"2026-07-13T22:20:51.137198Z","submitted_at":"2026-03-19T13:36:50Z","title":"Parallelizing Tool Execution and LLM Generation for Low-Latency Agent Serving","version":3},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-07-13T22:20:51.838309Z"},"links":{"citing_paper":"/paper/2603.18897"},"observation_digest":"sha256:3bbf87ed2e720b52b3d40dc6afced69435afc69b06b56ee0c08d2b1a8c98c65e","observation_id":"52834ad5-280f-4fc0-a5c2-808a4ba0e4bf","resolution":{"observed_at":"2026-07-13T22:20:51.838309Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.15332","last_updated":"2025-05-27T09:24:50Z","snapshot_observed_at":"2026-07-06T19:36:36.994783Z","submitted_at":"2024-10-20T08:42:29Z","title":"EPIC: Efficient Position-Independent Caching for Serving Large Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.15332","snapshot_observed_at":"2026-07-13T22:20:51.838309Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2603.18897","last_updated":"2026-06-16T09:34:16Z","snapshot_observed_at":"2026-07-13T22:20:51.137198Z","submitted_at":"2026-03-19T13:36:50Z","title":"Parallelizing Tool Execution and LLM Generation for Low-Latency Agent Serving","version":3},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-07-13T22:20:51.838309Z"},"links":{"cited_paper":"/paper/2410.15332","citing_paper":"/paper/2603.18897"},"observation_digest":"sha256:a6287a93baa60c29c2abcb7e1ebb0ce2651b0b2d37646acb91aa44719799b325","observation_id":"34912571-7245-46c1-b0b2-3bd55e22af32","resolution":{"observed_at":"2026-07-13T22:20:51.838309Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-07-13T22:20:51.838309Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2603.18897","last_updated":"2026-06-16T09:34:16Z","snapshot_observed_at":"2026-07-13T22:20:51.137198Z","submitted_at":"2026-03-19T13:36:50Z","title":"Parallelizing Tool Execution and LLM Generation for Low-Latency Agent Serving","version":3},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-07-13T22:20:51.838309Z"},"links":{"citing_paper":"/paper/2603.18897"},"observation_digest":"sha256:44dbda4b63f2fcc7f8a1fec26212b2a4ef03bed6197eddaefb5ce41e6f228409","observation_id":"59593e27-2022-4d89-a5af-23dbb5033123","resolution":{"observed_at":"2026-07-13T22:20:51.838309Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-07-13T22:20:51.838309Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2603.18897","last_updated":"2026-06-16T09:34:16Z","snapshot_observed_at":"2026-07-13T22:20:51.137198Z","submitted_at":"2026-03-19T13:36:50Z","title":"Parallelizing Tool Execution and LLM Generation for Low-Latency Agent Serving","version":3},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-07-13T22:20:51.838309Z"},"links":{"citing_paper":"/paper/2603.18897"},"observation_digest":"sha256:230a7236b47d9dd4ea9be05d93d7fc5561461003c0ea9899c9c4744571f53b89","observation_id":"89ef93ed-8452-4ef5-b939-96a6baf352b2","resolution":{"observed_at":"2026-07-13T22:20:51.838309Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-07-13T22:20:51.838309Z","title":"Abad, Gregory Van Seghbroeck, Sam Deckers, Alexander Lemmens, and Mohammad Shahrad","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2603.18897","last_updated":"2026-06-16T09:34:16Z","snapshot_observed_at":"2026-07-13T22:20:51.137198Z","submitted_at":"2026-03-19T13:36:50Z","title":"Parallelizing Tool Execution and LLM Generation for Low-Latency Agent Serving","version":3},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-07-13T22:20:51.838309Z"},"links":{"citing_paper":"/paper/2603.18897"},"observation_digest":"sha256:f8a5b13de9405c51c3c0357dae3c120b443bbac42943ab4706dfbb2a3bd9f139","observation_id":"b0ce1203-1d7d-4ff7-942d-fc08eda96665","resolution":{"observed_at":"2026-07-13T22:20:51.838309Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-07-13T22:20:51.838309Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2603.18897","last_updated":"2026-06-16T09:34:16Z","snapshot_observed_at":"2026-07-13T22:20:51.137198Z","submitted_at":"2026-03-19T13:36:50Z","title":"Parallelizing Tool Execution and LLM Generation for Low-Latency Agent Serving","version":3},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-07-13T22:20:51.838309Z"},"links":{"citing_paper":"/paper/2603.18897"},"observation_digest":"sha256:845717fde0f8702ca7ab6cbc2afa92c80d32bcce61a3288f34f668860c990e27","observation_id":"327316de-fd2f-4425-a7d3-7f63197fa029","resolution":{"observed_at":"2026-07-13T22:20:51.838309Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-07-13T22:20:51.838309Z","title":"Gonzalez, Hao Zhang, and Ion Stoica","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2603.18897","last_updated":"2026-06-16T09:34:16Z","snapshot_observed_at":"2026-07-13T22:20:51.137198Z","submitted_at":"2026-03-19T13:36:50Z","title":"Parallelizing Tool Execution and LLM Generation for Low-Latency Agent Serving","version":3},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-07-13T22:20:51.838309Z"},"links":{"citing_paper":"/paper/2603.18897"},"observation_digest":"sha256:63957bd4677fa18dcf68079808a087e22c92cac85836de47875bd79bdb8ca022","observation_id":"9bc6a2f6-6a74-46e3-83ab-eee15055f1e5","resolution":{"observed_at":"2026-07-13T22:20:51.838309Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2504.07347","last_updated":"2026-05-18T01:04:46Z","snapshot_observed_at":"2026-08-04T04:26:27.630303Z","submitted_at":"2025-04-10T00:12:12Z","title":"Throughput-Optimal Scheduling Algorithms for LLM Inference and AI Agents","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.07347","snapshot_observed_at":"2026-07-13T22:20:51.838309Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2603.18897","last_updated":"2026-06-16T09:34:16Z","snapshot_observed_at":"2026-07-13T22:20:51.137198Z","submitted_at":"2026-03-19T13:36:50Z","title":"Parallelizing Tool Execution and LLM Generation for Low-Latency Agent Serving","version":3},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-07-13T22:20:51.838309Z"},"links":{"cited_paper":"/paper/2504.07347","citing_paper":"/paper/2603.18897"},"observation_digest":"sha256:53ba841b37b89d3155ee9ac6516cac9603cda49ee92693ca8d2e6a246f7b2cc1","observation_id":"fecb0bf2-c466-4f06-8ba1-d5db71a87279","resolution":{"observed_at":"2026-07-13T22:20:51.838309Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-07-13T22:20:51.838309Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2603.18897","last_updated":"2026-06-16T09:34:16Z","snapshot_observed_at":"2026-07-13T22:20:51.137198Z","submitted_at":"2026-03-19T13:36:50Z","title":"Parallelizing Tool Execution and LLM Generation for Low-Latency Agent Serving","version":3},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-07-13T22:20:51.838309Z"},"links":{"citing_paper":"/paper/2603.18897"},"observation_digest":"sha256:be5c833336a1c303846c1ff90d2244f8b284d8a103b53faf8b6c403714559956","observation_id":"afff0986-e5ce-4e9a-a9c4-679069af122c","resolution":{"observed_at":"2026-07-13T22:20:51.838309Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-07-13T22:20:51.838309Z","title":"Liu, Amit Levy, Shadi Noghabi, and Sebastian Burckhardt","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2603.18897","last_updated":"2026-06-16T09:34:16Z","snapshot_observed_at":"2026-07-13T22:20:51.137198Z","submitted_at":"2026-03-19T13:36:50Z","title":"Parallelizing Tool Execution and LLM Generation for Low-Latency Agent Serving","version":3},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-07-13T22:20:51.838309Z"},"links":{"citing_paper":"/paper/2603.18897"},"observation_digest":"sha256:2f8a3cb270f69e952b5275116348e2e916f5d855d9b3b13ce4f2afa6583f123d","observation_id":"4e746e40-8b5f-46ea-b415-03f3b2edce09","resolution":{"observed_at":"2026-07-13T22:20:51.838309Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-07-13T22:20:51.838309Z","title":"Larus, and Haibo Chen","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2603.18897","last_updated":"2026-06-16T09:34:16Z","snapshot_observed_at":"2026-07-13T22:20:51.137198Z","submitted_at":"2026-03-19T13:36:50Z","title":"Parallelizing Tool Execution and LLM Generation for Low-Latency Agent Serving","version":3},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-07-13T22:20:51.838309Z"},"links":{"citing_paper":"/paper/2603.18897"},"observation_digest":"sha256:f88f83446a8d13798730e048c83abd3214c229650869f74e23bd490e839d0436","observation_id":"6dcbf4e5-b17b-49b4-87cb-4f502246a9fe","resolution":{"observed_at":"2026-07-13T22:20:51.838309Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.13965","last_updated":"2025-02-19T18:59:30Z","snapshot_observed_at":"2026-07-06T20:39:19.627410Z","submitted_at":"2025-02-19T18:59:30Z","title":"Autellix: An Efficient Serving Engine for LLM Agents as General Programs","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.13965","snapshot_observed_at":"2026-07-13T22:20:51.838309Z","title":"Gonzalez, and Ion Stoica","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2603.18897","last_updated":"2026-06-16T09:34:16Z","snapshot_observed_at":"2026-07-13T22:20:51.137198Z","submitted_at":"2026-03-19T13:36:50Z","title":"Parallelizing Tool Execution and LLM Generation for Low-Latency Agent Serving","version":3},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-07-13T22:20:51.838309Z"},"links":{"cited_paper":"/paper/2502.13965","citing_paper":"/paper/2603.18897"},"observation_digest":"sha256:176640ce3fb3c8d08b20212e9feaba16dc16415b149517027a12ffc03f6aa180","observation_id":"45b588fb-3348-46f4-952e-ffb6489af2bf","resolution":{"observed_at":"2026-07-13T22:20:51.838309Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-07-13T22:20:51.838309Z","title":null,"venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2603.18897","last_updated":"2026-06-16T09:34:16Z","snapshot_observed_at":"2026-07-13T22:20:51.137198Z","submitted_at":"2026-03-19T13:36:50Z","title":"Parallelizing Tool Execution and LLM Generation for Low-Latency Agent Serving","version":3},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-07-13T22:20:51.838309Z"},"links":{"citing_paper":"/paper/2603.18897"},"observation_digest":"sha256:8c1d13682ca4aadf422b594790098eade6a4b94f6195a8bddd87feeaba8ca21d","observation_id":"28ac6e9e-c45e-4be9-ac71-3a7b0d35c947","resolution":{"observed_at":"2026-07-13T22:20:51.838309Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-07-13T22:20:51.838309Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2603.18897","last_updated":"2026-06-16T09:34:16Z","snapshot_observed_at":"2026-07-13T22:20:51.137198Z","submitted_at":"2026-03-19T13:36:50Z","title":"Parallelizing Tool Execution and LLM Generation for Low-Latency Agent Serving","version":3},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-07-13T22:20:51.838309Z"},"links":{"citing_paper":"/paper/2603.18897"},"observation_digest":"sha256:0611ddeffc99799e334dabdd0421aa0d394b3c0097669f269e3194dfcf266ee9","observation_id":"b62924f8-e902-43e9-bb05-06deb7fe032e","resolution":{"observed_at":"2026-07-13T22:20:51.838309Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2508.15351","last_updated":"2025-08-26T12:14:11Z","snapshot_observed_at":"2026-08-05T17:57:34.885579Z","submitted_at":"2025-08-21T08:29:01Z","title":"Databelt: A Continuous Data Path for Serverless Workflows in the 3D Compute Continuum","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2508.15351","snapshot_observed_at":"2026-07-13T22:20:51.838309Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2603.18897","last_updated":"2026-06-16T09:34:16Z","snapshot_observed_at":"2026-07-13T22:20:51.137198Z","submitted_at":"2026-03-19T13:36:50Z","title":"Parallelizing Tool Execution and LLM Generation for Low-Latency Agent Serving","version":3},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-07-13T22:20:51.838309Z"},"links":{"cited_paper":"/paper/2508.15351","citing_paper":"/paper/2603.18897"},"observation_digest":"sha256:e7a054b88c200301e782e78246ece0fa3c9e2f8672961f49dbaed4555a081f87","observation_id":"0a7e8d51-0034-4ec9-8d4d-69898f0158b4","resolution":{"observed_at":"2026-07-13T22:20:51.838309Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"3386.2024","doi":"10.1109/ucc63386.2024.00017","metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":null,"venue":null,"work_id":"6cda1526-c9c9-428c-bb5a-2511c79f066f","year":2024},"citing_paper":{"arxiv_id":"2603.18897","last_updated":"2026-06-16T09:34:16Z","snapshot_observed_at":"2026-07-13T22:20:51.137198Z","submitted_at":"2026-03-19T13:36:50Z","title":"Parallelizing Tool Execution and LLM Generation for Low-Latency Agent Serving","version":3},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-07-13T22:20:51.838309Z"},"links":{"citing_paper":"/paper/2603.18897"},"observation_digest":"sha256:b26b9a5edbee10adefff860a6e44857552c961d88cc64f82e67d57433ced0316","observation_id":"bcb091e3-641f-496a-8a18-c1d91548203f","resolution":{"observed_at":"2026-07-13T22:29:58.178209Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-05T06:32:48.257954+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-07-13T22:20:51.838309Z","title":"2025.Kimi-Researcher: End-to-End RL Training for Emerging Agentic Capabilities","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2603.18897","last_updated":"2026-06-16T09:34:16Z","snapshot_observed_at":"2026-07-13T22:20:51.137198Z","submitted_at":"2026-03-19T13:36:50Z","title":"Parallelizing Tool Execution and LLM Generation for Low-Latency Agent Serving","version":3},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-07-13T22:20:51.838309Z"},"links":{"citing_paper":"/paper/2603.18897"},"observation_digest":"sha256:5c7b45df7be5d90bd74f9ec169d10e79147be201ac7768824d13cce06ae20f58","observation_id":"e299d80f-139b-4a8c-834f-da2120e842f5","resolution":{"observed_at":"2026-07-13T22:20:51.838309Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-07-13T22:20:51.838309Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2603.18897","last_updated":"2026-06-16T09:34:16Z","snapshot_observed_at":"2026-07-13T22:20:51.137198Z","submitted_at":"2026-03-19T13:36:50Z","title":"Parallelizing Tool Execution and LLM Generation for Low-Latency Agent Serving","version":3},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-07-13T22:20:51.838309Z"},"links":{"citing_paper":"/paper/2603.18897"},"observation_digest":"sha256:37b889a648bb2e6f13f49cf27c78e9c34ad64eeb0b5999576efe376359c1a2c3","observation_id":"b319c062-016f-4349-8993-1539bf5e008b","resolution":{"observed_at":"2026-07-13T22:20:51.838309Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-07-13T22:20:51.838309Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2603.18897","last_updated":"2026-06-16T09:34:16Z","snapshot_observed_at":"2026-07-13T22:20:51.137198Z","submitted_at":"2026-03-19T13:36:50Z","title":"Parallelizing Tool Execution and LLM Generation for Low-Latency Agent Serving","version":3},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-07-13T22:20:51.838309Z"},"links":{"citing_paper":"/paper/2603.18897"},"observation_digest":"sha256:906702647507e0693d5d24b5ba1cf4a6b99fac8dec4840c24b940cd8026e7bf4","observation_id":"bd7b1c2e-8126-45f1-a497-f80ee8ee7513","resolution":{"observed_at":"2026-07-13T22:20:51.838309Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2507.07400","last_updated":"2025-07-10T03:39:23Z","snapshot_observed_at":"2026-07-06T21:54:52.797028Z","submitted_at":"2025-07-10T03:39:23Z","title":"KVFlow: Efficient Prefix Caching for Accelerating LLM-Based Multi-Agent Workflows","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2507.07400","snapshot_observed_at":"2026-07-13T22:20:51.838309Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2603.18897","last_updated":"2026-06-16T09:34:16Z","snapshot_observed_at":"2026-07-13T22:20:51.137198Z","submitted_at":"2026-03-19T13:36:50Z","title":"Parallelizing Tool Execution and LLM Generation for Low-Latency Agent Serving","version":3},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-07-13T22:20:51.838309Z"},"links":{"cited_paper":"/paper/2507.07400","citing_paper":"/paper/2603.18897"},"observation_digest":"sha256:e2924cb69023c37c122c4e3bc0aa2bb4a72847d1057b6b2190b2dfb3e1dd9fcc","observation_id":"26d3d1de-fff5-4a5c-9e1a-38e9331db713","resolution":{"observed_at":"2026-07-13T22:20:51.838309Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2405.04437","last_updated":"2025-01-29T04:10:41Z","snapshot_observed_at":"2026-08-05T02:13:00.249260Z","submitted_at":"2024-05-07T16:00:32Z","title":"vAttention: Dynamic Memory Management for Serving LLMs without PagedAttention","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2405.04437","snapshot_observed_at":"2026-07-13T22:20:51.838309Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2603.18897","last_updated":"2026-06-16T09:34:16Z","snapshot_observed_at":"2026-07-13T22:20:51.137198Z","submitted_at":"2026-03-19T13:36:50Z","title":"Parallelizing Tool Execution and LLM Generation for Low-Latency Agent Serving","version":3},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-07-13T22:20:51.838309Z"},"links":{"cited_paper":"/paper/2405.04437","citing_paper":"/paper/2603.18897"},"observation_digest":"sha256:179f3fddfda2c8e2e9fb87b9ed40a0f7fcc8eb3394fc46b66f835903c7c34ff2","observation_id":"0f85f268-14ee-492e-93a6-fd93f1c5731b","resolution":{"observed_at":"2026-07-13T22:20:51.838309Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.01228","last_updated":"2025-09-03T20:54:57Z","snapshot_observed_at":"2026-07-06T19:25:42.156795Z","submitted_at":"2024-10-02T04:12:13Z","title":"ConServe: Fine-Grained GPU Harvesting for LLM Online and Offline Co-Serving","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.01228","snapshot_observed_at":"2026-07-13T22:20:51.838309Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2603.18897","last_updated":"2026-06-16T09:34:16Z","snapshot_observed_at":"2026-07-13T22:20:51.137198Z","submitted_at":"2026-03-19T13:36:50Z","title":"Parallelizing Tool Execution and LLM Generation for Low-Latency Agent Serving","version":3},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-07-13T22:20:51.838309Z"},"links":{"cited_paper":"/paper/2410.01228","citing_paper":"/paper/2603.18897"},"observation_digest":"sha256:a808774caa53435f92d6309b6745b62b351b5caedca6a2beda70c006e28f8b2f","observation_id":"62162d4d-2118-4406-a6b9-2c6519d98415","resolution":{"observed_at":"2026-07-13T22:20:51.838309Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-07-13T22:20:51.838309Z","title":"Kalbarczyk, Tamer Baçar, and Ravishankar K","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2603.18897","last_updated":"2026-06-16T09:34:16Z","snapshot_observed_at":"2026-07-13T22:20:51.137198Z","submitted_at":"2026-03-19T13:36:50Z","title":"Parallelizing Tool Execution and LLM Generation for Low-Latency Agent Serving","version":3},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-07-13T22:20:51.838309Z"},"links":{"citing_paper":"/paper/2603.18897"},"observation_digest":"sha256:284b38d035748a87bfbc01cc8c32de422e1fe4910c0ad86fb6694c8d7688ec56","observation_id":"1b08414a-659c-4822-8d82-fdab6cb03b54","resolution":{"observed_at":"2026-07-13T22:20:51.838309Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.08509","last_updated":"2024-11-25T17:35:07Z","snapshot_observed_at":"2026-07-06T17:59:25.200363Z","submitted_at":"2024-04-12T14:46:15Z","title":"Efficient Interactive LLM Serving with Proxy Model-based Sequence Length Prediction","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.08509","snapshot_observed_at":"2026-07-13T22:20:51.838309Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2603.18897","last_updated":"2026-06-16T09:34:16Z","snapshot_observed_at":"2026-07-13T22:20:51.137198Z","submitted_at":"2026-03-19T13:36:50Z","title":"Parallelizing Tool Execution and LLM Generation for Low-Latency Agent Serving","version":3},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-07-13T22:20:51.838309Z"},"links":{"cited_paper":"/paper/2404.08509","citing_paper":"/paper/2603.18897"},"observation_digest":"sha256:480bbdd21f657e7b987a72def92a804e729b8e52c390f5162b595da6cc3c67af","observation_id":"480d42c1-e30b-4030-9696-52950dec518b","resolution":{"observed_at":"2026-07-13T22:20:51.838309Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-07-13T22:20:51.838309Z","title":null,"venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2603.18897","last_updated":"2026-06-16T09:34:16Z","snapshot_observed_at":"2026-07-13T22:20:51.137198Z","submitted_at":"2026-03-19T13:36:50Z","title":"Parallelizing Tool Execution and LLM Generation for Low-Latency Agent Serving","version":3},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-07-13T22:20:51.838309Z"},"links":{"citing_paper":"/paper/2603.18897"},"observation_digest":"sha256:ba076d0ea201fe251361a47c0de08233b10044e035dc0c2f728488342f6b4761","observation_id":"49624768-70f9-44f1-82a5-1a6d9ad87fc2","resolution":{"observed_at":"2026-07-13T22:20:51.838309Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-07-13T22:20:51.838309Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2603.18897","last_updated":"2026-06-16T09:34:16Z","snapshot_observed_at":"2026-07-13T22:20:51.137198Z","submitted_at":"2026-03-19T13:36:50Z","title":"Parallelizing Tool Execution and LLM Generation for Low-Latency Agent Serving","version":3},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-07-13T22:20:51.838309Z"},"links":{"citing_paper":"/paper/2603.18897"},"observation_digest":"sha256:49f389f7e6a80bcd44bba6d80a3beb063fd69ce059def48f247d342ea5f30168","observation_id":"1e8edd2c-88db-4e3e-8fb7-8f28c35536cb","resolution":{"observed_at":"2026-07-13T22:20:51.838309Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-07-13T22:20:51.838309Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2603.18897","last_updated":"2026-06-16T09:34:16Z","snapshot_observed_at":"2026-07-13T22:20:51.137198Z","submitted_at":"2026-03-19T13:36:50Z","title":"Parallelizing Tool Execution and LLM Generation for Low-Latency Agent Serving","version":3},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-07-13T22:20:51.838309Z"},"links":{"citing_paper":"/paper/2603.18897"},"observation_digest":"sha256:49ad57c31290702f5a229dfd7704d19522a3de23b39b16c6198e732b1477d82b","observation_id":"64366b63-144c-4954-9dc1-34d40e8d5059","resolution":{"observed_at":"2026-07-13T22:20:51.838309Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-07-13T22:20:51.838309Z","title":null,"venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2603.18897","last_updated":"2026-06-16T09:34:16Z","snapshot_observed_at":"2026-07-13T22:20:51.137198Z","submitted_at":"2026-03-19T13:36:50Z","title":"Parallelizing Tool Execution and LLM Generation for Low-Latency Agent Serving","version":3},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-07-13T22:20:51.838309Z"},"links":{"citing_paper":"/paper/2603.18897"},"observation_digest":"sha256:6e92a1d92e562c2978c37e182c220421dede490581165570eec5d519c4974d4f","observation_id":"dcd7f1ed-5270-4a36-9107-b3694298d29e","resolution":{"observed_at":"2026-07-13T22:20:51.838309Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-07-13T22:20:51.838309Z","title":"Bulaong, John E","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2603.18897","last_updated":"2026-06-16T09:34:16Z","snapshot_observed_at":"2026-07-13T22:20:51.137198Z","submitted_at":"2026-03-19T13:36:50Z","title":"Parallelizing Tool Execution and LLM Generation for Low-Latency Agent Serving","version":3},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-07-13T22:20:51.838309Z"},"links":{"citing_paper":"/paper/2603.18897"},"observation_digest":"sha256:c2f4c4a4a3373bf39b4ec22da8e2fd7caf2bdd7100cb79cbcfee72e14ff08489","observation_id":"253ed573-b5ff-4e2a-a704-579b265770f0","resolution":{"observed_at":"2026-07-13T22:20:51.838309Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-07-13T22:20:51.838309Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2603.18897","last_updated":"2026-06-16T09:34:16Z","snapshot_observed_at":"2026-07-13T22:20:51.137198Z","submitted_at":"2026-03-19T13:36:50Z","title":"Parallelizing Tool Execution and LLM Generation for Low-Latency Agent Serving","version":3},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-07-13T22:20:51.838309Z"},"links":{"citing_paper":"/paper/2603.18897"},"observation_digest":"sha256:024146b5eed1fb3862ef979a7986bc6a281034dac51dd6bbbbc78f1654279827","observation_id":"bc394fa0-f377-4246-88a0-35c8a1323a94","resolution":{"observed_at":"2026-07-13T22:20:51.838309Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-07-13T22:20:51.838309Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2603.18897","last_updated":"2026-06-16T09:34:16Z","snapshot_observed_at":"2026-07-13T22:20:51.137198Z","submitted_at":"2026-03-19T13:36:50Z","title":"Parallelizing Tool Execution and LLM Generation for Low-Latency Agent Serving","version":3},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-07-13T22:20:51.838309Z"},"links":{"citing_paper":"/paper/2603.18897"},"observation_digest":"sha256:4e78e315c04b6f6bb7ff6209ae1be2b4a3f93da4d0807fa2b0cf0342aa463717","observation_id":"b0c949cf-11f4-4ac2-b4ae-9f504d64c427","resolution":{"observed_at":"2026-07-13T22:20:51.838309Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2512.04013","last_updated":"2026-07-10T15:00:48Z","snapshot_observed_at":"2026-08-03T18:43:26.228239Z","submitted_at":"2025-12-03T17:49:38Z","title":"AugServe: Adaptive Request Scheduling for Augmented Large Language Model Inference Serving","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2512.04013","snapshot_observed_at":"2026-07-13T22:20:51.838309Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2603.18897","last_updated":"2026-06-16T09:34:16Z","snapshot_observed_at":"2026-07-13T22:20:51.137198Z","submitted_at":"2026-03-19T13:36:50Z","title":"Parallelizing Tool Execution and LLM Generation for Low-Latency Agent Serving","version":3},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-07-13T22:20:51.838309Z"},"links":{"cited_paper":"/paper/2512.04013","citing_paper":"/paper/2603.18897"},"observation_digest":"sha256:84b686b16e5eb90aaca5ec1a571819a1c6a03dbbccdb2ea0d5c6a61c9fbc4811","observation_id":"e9d5cf38-73a0-406b-97d0-83e3602788ed","resolution":{"observed_at":"2026-07-13T22:20:51.838309Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.05920","last_updated":"2024-09-25T05:57:51Z","snapshot_observed_at":"2026-07-06T15:25:24.491136Z","submitted_at":"2023-05-10T06:17:50Z","title":"Fast Distributed Inference Serving for Large Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.05920","snapshot_observed_at":"2026-07-13T22:20:51.838309Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2603.18897","last_updated":"2026-06-16T09:34:16Z","snapshot_observed_at":"2026-07-13T22:20:51.137198Z","submitted_at":"2026-03-19T13:36:50Z","title":"Parallelizing Tool Execution and LLM Generation for Low-Latency Agent Serving","version":3},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-07-13T22:20:51.838309Z"},"links":{"cited_paper":"/paper/2305.05920","citing_paper":"/paper/2603.18897"},"observation_digest":"sha256:ce362182976b9b635f1a9a15674be6c82d1e86dc65052e323975c6b8e35f2300","observation_id":"91b88b95-dd94-4824-851c-9f5e95cf7fa6","resolution":{"observed_at":"2026-07-13T22:20:51.838309Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.01830","last_updated":"2024-11-04T06:04:07Z","snapshot_observed_at":"2026-07-06T19:44:38.954993Z","submitted_at":"2024-11-04T06:04:07Z","title":"FaaSTube: Optimizing GPU-oriented Data Transfer for Serverless Computing","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.01830","snapshot_observed_at":"2026-07-13T22:20:51.838309Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2603.18897","last_updated":"2026-06-16T09:34:16Z","snapshot_observed_at":"2026-07-13T22:20:51.137198Z","submitted_at":"2026-03-19T13:36:50Z","title":"Parallelizing Tool Execution and LLM Generation for Low-Latency Agent Serving","version":3},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-07-13T22:20:51.838309Z"},"links":{"cited_paper":"/paper/2411.01830","citing_paper":"/paper/2603.18897"},"observation_digest":"sha256:161ff79da6a1d9b051a95fa4dbaa5eee10fbc6db166ca5295f7c6f6d02222ad2","observation_id":"406f9a7a-3c6f-4946-acb1-a60b80658762","resolution":{"observed_at":"2026-07-13T22:20:51.838309Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2508.18572","last_updated":"2025-08-26T00:09:03Z","snapshot_observed_at":"2026-08-05T16:20:58.193748Z","submitted_at":"2025-08-26T00:09:03Z","title":"Strata: Hierarchical Context Caching for Long Context Language Model Serving","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2508.18572","snapshot_observed_at":"2026-07-13T22:20:51.838309Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2603.18897","last_updated":"2026-06-16T09:34:16Z","snapshot_observed_at":"2026-07-13T22:20:51.137198Z","submitted_at":"2026-03-19T13:36:50Z","title":"Parallelizing Tool Execution and LLM Generation for Low-Latency Agent Serving","version":3},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-07-13T22:20:51.838309Z"},"links":{"cited_paper":"/paper/2508.18572","citing_paper":"/paper/2603.18897"},"observation_digest":"sha256:9761070c9757a7e9e2d1bc7b53d1d7abe2fb878b72c40d564781100ad694ecec","observation_id":"f0f091cc-6658-4be0-a16b-41d239073977","resolution":{"observed_at":"2026-07-13T22:20:51.838309Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-07-13T22:20:51.838309Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2603.18897","last_updated":"2026-06-16T09:34:16Z","snapshot_observed_at":"2026-07-13T22:20:51.137198Z","submitted_at":"2026-03-19T13:36:50Z","title":"Parallelizing Tool Execution and LLM Generation for Low-Latency Agent Serving","version":3},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-07-13T22:20:51.838309Z"},"links":{"citing_paper":"/paper/2603.18897"},"observation_digest":"sha256:e56ae8c50a5c9cdcf95100dfe52d6b2c2921b7e0d1103441e30d154cc2a4fe96","observation_id":"a9b187b8-39ed-4523-84dd-63cc210e83ea","resolution":{"observed_at":"2026-07-13T22:20:51.838309Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-07-13T22:20:51.838309Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2603.18897","last_updated":"2026-06-16T09:34:16Z","snapshot_observed_at":"2026-07-13T22:20:51.137198Z","submitted_at":"2026-03-19T13:36:50Z","title":"Parallelizing Tool Execution and LLM Generation for Low-Latency Agent Serving","version":3},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-07-13T22:20:51.838309Z"},"links":{"citing_paper":"/paper/2603.18897"},"observation_digest":"sha256:baef05a294394118e7da543a180985742789c9daea70b08375355a2ebd6485c8","observation_id":"dab70be0-a06b-4e76-a3f1-8a5fb78bdbb8","resolution":{"observed_at":"2026-07-13T22:20:51.838309Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2510.04371","last_updated":"2026-04-23T17:58:32Z","snapshot_observed_at":"2026-07-06T22:31:44.964774Z","submitted_at":"2025-10-05T21:28:11Z","title":"Speculative Actions: A Lossless Framework for Faster Agentic Systems","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2510.04371","snapshot_observed_at":"2026-07-13T22:20:51.838309Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2603.18897","last_updated":"2026-06-16T09:34:16Z","snapshot_observed_at":"2026-07-13T22:20:51.137198Z","submitted_at":"2026-03-19T13:36:50Z","title":"Parallelizing Tool Execution and LLM Generation for Low-Latency Agent Serving","version":3},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-07-13T22:20:51.838309Z"},"links":{"cited_paper":"/paper/2510.04371","citing_paper":"/paper/2603.18897"},"observation_digest":"sha256:8a97b87becef6339551add2cae02fc14e04fd3738f3e3da705f70dcf4aa16a46","observation_id":"384768e5-762b-4aad-9470-f6e07c411a30","resolution":{"observed_at":"2026-07-13T22:20:51.838309Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-07-13T22:20:51.838309Z","title":null,"venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2603.18897","last_updated":"2026-06-16T09:34:16Z","snapshot_observed_at":"2026-07-13T22:20:51.137198Z","submitted_at":"2026-03-19T13:36:50Z","title":"Parallelizing Tool Execution and LLM Generation for Low-Latency Agent Serving","version":3},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-07-13T22:20:51.838309Z"},"links":{"citing_paper":"/paper/2603.18897"},"observation_digest":"sha256:f4648fae3acff5d42a016b2901f0b8744b4f25365ac2e7340d957ab5428082f7","observation_id":"5e41df4b-f2da-4ebd-90c6-1f9af5231f88","resolution":{"observed_at":"2026-07-13T22:20:51.838309Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-07-13T22:20:51.838309Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2603.18897","last_updated":"2026-06-16T09:34:16Z","snapshot_observed_at":"2026-07-13T22:20:51.137198Z","submitted_at":"2026-03-19T13:36:50Z","title":"Parallelizing Tool Execution and LLM Generation for Low-Latency Agent Serving","version":3},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-07-13T22:20:51.838309Z"},"links":{"citing_paper":"/paper/2603.18897"},"observation_digest":"sha256:8950c70261860f4cc926a1496841d121f66ea55a3852735da8f7f5b8d1a44ab9","observation_id":"739cb450-2b20-42c8-a193-37c1213e08f5","resolution":{"observed_at":"2026-07-13T22:20:51.838309Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-07-13T22:20:51.838309Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2603.18897","last_updated":"2026-06-16T09:34:16Z","snapshot_observed_at":"2026-07-13T22:20:51.137198Z","submitted_at":"2026-03-19T13:36:50Z","title":"Parallelizing Tool Execution and LLM Generation for Low-Latency Agent Serving","version":3},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-07-13T22:20:51.838309Z"},"links":{"citing_paper":"/paper/2603.18897"},"observation_digest":"sha256:6fe293755da270f49c2caad78a0203187ce930284943a4e1a099f2de426e59d0","observation_id":"8cf788e3-fc53-49fd-8857-71227f2c911a","resolution":{"observed_at":"2026-07-13T22:20:51.838309Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-07-13T22:20:51.838309Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2603.18897","last_updated":"2026-06-16T09:34:16Z","snapshot_observed_at":"2026-07-13T22:20:51.137198Z","submitted_at":"2026-03-19T13:36:50Z","title":"Parallelizing Tool Execution and LLM Generation for Low-Latency Agent Serving","version":3},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-07-13T22:20:51.838309Z"},"links":{"citing_paper":"/paper/2603.18897"},"observation_digest":"sha256:a370aa7c3bf53b05758a2274f5f45cf1e09a1a73d1a47380542cf5c7a4cb0547","observation_id":"2dc539a5-b12b-4c2f-aaad-c2af6c1f5ff3","resolution":{"observed_at":"2026-07-13T22:20:51.838309Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-07-13T22:20:51.838309Z","title":null,"venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2603.18897","last_updated":"2026-06-16T09:34:16Z","snapshot_observed_at":"2026-07-13T22:20:51.137198Z","submitted_at":"2026-03-19T13:36:50Z","title":"Parallelizing Tool Execution and LLM Generation for Low-Latency Agent Serving","version":3},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-07-13T22:20:51.838309Z"},"links":{"citing_paper":"/paper/2603.18897"},"observation_digest":"sha256:9cc15a63a9b1062478847021f7de194c4bbaae8bb9fd12ebdb70e5534e1da22a","observation_id":"024929c1-56a8-4cef-a3ef-5293319fde6e","resolution":{"observed_at":"2026-07-13T22:20:51.838309Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-07-13T22:20:51.838309Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2603.18897","last_updated":"2026-06-16T09:34:16Z","snapshot_observed_at":"2026-07-13T22:20:51.137198Z","submitted_at":"2026-03-19T13:36:50Z","title":"Parallelizing Tool Execution and LLM Generation for Low-Latency Agent Serving","version":3},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-07-13T22:20:51.838309Z"},"links":{"citing_paper":"/paper/2603.18897"},"observation_digest":"sha256:0071f37e9029eb9dbc925f812c65b01878e95ca805c05c56e266eb7312a6f808","observation_id":"1af88d74-17ff-4dfe-b4ef-289be57d7f65","resolution":{"observed_at":"2026-07-13T22:20:51.838309Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2603.18897","last_updated":"2026-06-16T09:34:16Z","latest_version":3,"primary_category":"cs.DC","snapshot_observed_at":"2026-07-13T22:20:51.137198Z","submitted_at":"2026-03-19T13:36:50Z","title":"Parallelizing Tool Execution and LLM Generation for Low-Latency Agent Serving"},"reference_resolution":{"displayed":60,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":59,"verified_exact":1,"verified_fuzzy":0},"total_outbound_references":60},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-05T06:32:48.257954+00:00","source":"crossref"},{"observed_at":"2026-08-05T06:32:44.755628+00:00","source":"retraction_watch"}],"thesis":"As of 6 August 2026, this Paper Citation Record lists 60 of 60 outbound references and 11 inbound Pith citation observations for arXiv:2603.18897."}