{"as_of":"2026-08-10T00:25:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:f766d76336e1ae127ad2e9b7f5d34212fa00c229af471f43c059b935aabfe408","coverage":[{"denominator":25,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":25,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-01T01:35:09.720250Z","state":"measured"},{"denominator":25,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":25,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-09T06:31:02.800959+00:00","state":"measured"},{"denominator":0,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"cited_works","source_observed_at":null,"state":"measured"}],"external_citation_measurements":[],"inbound":[],"links":{"evidence":"/evidence","html":"/paper/2607.25765/citation-record","integrity":"/paper/2607.25765/integrity","json":"/paper/2607.25765/citation-record.json","paper":"/paper/2607.25765"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T01:35:09.614992Z","title":"Hybridqa: A dataset of multi-hop question answering over tabular and textual data","venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2607.25765","last_updated":"2026-07-28T14:19:59Z","snapshot_observed_at":"2026-08-09T00:29:07.819944Z","submitted_at":"2026-07-28T14:19:59Z","title":"WorkSurface-Bench: Benchmarking Enterprise Agents on Multi-Surface Knowledge Routing","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-01T01:35:09.614992Z"},"links":{"citing_paper":"/paper/2607.25765"},"observation_digest":"sha256:d672ba22fb167828f835ede2a605c9eddc9f84ea1e497617467d26de6383ae28","observation_id":"97c3c300-2a7b-4155-89fb-9a6cb60c60bb","resolution":{"observed_at":"2026-08-01T01:35:09.614992Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T01:35:09.620361Z","title":"Open question answering over tables and text","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2607.25765","last_updated":"2026-07-28T14:19:59Z","snapshot_observed_at":"2026-08-09T00:29:07.819944Z","submitted_at":"2026-07-28T14:19:59Z","title":"WorkSurface-Bench: Benchmarking Enterprise Agents on Multi-Surface Knowledge Routing","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-01T01:35:09.620361Z"},"links":{"citing_paper":"/paper/2607.25765"},"observation_digest":"sha256:8f5703ba62ccd26897fefc96d87f5448006333286d8019f57d4e4ae1a6367876","observation_id":"70c16903-b8fd-4ab8-b124-f91dc08333ec","resolution":{"observed_at":"2026-08-01T01:35:09.620361Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.14033","last_updated":"2024-01-15T03:18:25Z","snapshot_observed_at":"2026-07-06T17:06:37.369542Z","submitted_at":"2023-12-21T17:02:06Z","title":"T-Eval: Evaluating the Tool Utilization Capability of Large Language Models Step by Step","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.14033","snapshot_observed_at":"2026-08-01T01:35:09.624996Z","title":"T-eval: Evaluating the tool utilization capability of large language models step by step.arXiv preprint arXiv:2312.14033, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.25765","last_updated":"2026-07-28T14:19:59Z","snapshot_observed_at":"2026-08-09T00:29:07.819944Z","submitted_at":"2026-07-28T14:19:59Z","title":"WorkSurface-Bench: Benchmarking Enterprise Agents on Multi-Surface Knowledge Routing","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-01T01:35:09.624996Z"},"links":{"cited_paper":"/paper/2312.14033","citing_paper":"/paper/2607.25765"},"observation_digest":"sha256:a92d913dce099440d8b6a11d206a64b5f697835de0517e1c6ce1dc625a604daa","observation_id":"ee8046a5-274a-4651-92ce-9272532307be","resolution":{"observed_at":"2026-08-01T01:35:09.624996Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2506.23139","last_updated":"2025-06-29T08:34:59Z","snapshot_observed_at":"2026-08-07T20:29:49.293010Z","submitted_at":"2025-06-29T08:34:59Z","title":"Benchmarking Deep Search over Heterogeneous Enterprise Data","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2506.23139","snapshot_observed_at":"2026-08-01T01:35:09.630285Z","title":"Benchmarking deep search over heterogeneous enterprise data.arXiv preprint arXiv:2506.23139, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.25765","last_updated":"2026-07-28T14:19:59Z","snapshot_observed_at":"2026-08-09T00:29:07.819944Z","submitted_at":"2026-07-28T14:19:59Z","title":"WorkSurface-Bench: Benchmarking Enterprise Agents on Multi-Surface Knowledge Routing","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-01T01:35:09.630285Z"},"links":{"cited_paper":"/paper/2506.23139","citing_paper":"/paper/2607.25765"},"observation_digest":"sha256:5111bc99525b5aa13eb7155ff8d82019477cc56a46407d9b099e1c280e80c0d5","observation_id":"2e2579c7-758e-4e43-96c5-50ca9ab3faf5","resolution":{"observed_at":"2026-08-01T01:35:09.630285Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T01:35:09.635073Z","title":"DeepSeek V4 Preview Release","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2607.25765","last_updated":"2026-07-28T14:19:59Z","snapshot_observed_at":"2026-08-09T00:29:07.819944Z","submitted_at":"2026-07-28T14:19:59Z","title":"WorkSurface-Bench: Benchmarking Enterprise Agents on Multi-Surface Knowledge Routing","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-01T01:35:09.635073Z"},"links":{"citing_paper":"/paper/2607.25765"},"observation_digest":"sha256:1c67b5e974d0edb1d2be442f8ef3ff9d721aa7939134e2c9d91ab857527ccb9c","observation_id":"249c97c0-f652-4089-9d24-fe6194ac8921","resolution":{"observed_at":"2026-08-01T01:35:09.635073Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2403.07718","last_updated":"2024-07-23T06:19:28Z","snapshot_observed_at":"2026-08-02T12:09:24.340284Z","submitted_at":"2024-03-12T14:58:45Z","title":"WorkArena: How Capable Are Web Agents at Solving Common Knowledge Work Tasks?","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2403.07718","snapshot_observed_at":"2026-08-01T01:35:09.639378Z","title":"Workarena: How capable are web agents at solving common knowledge work tasks?arXiv preprint arXiv:2403.07718, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.25765","last_updated":"2026-07-28T14:19:59Z","snapshot_observed_at":"2026-08-09T00:29:07.819944Z","submitted_at":"2026-07-28T14:19:59Z","title":"WorkSurface-Bench: Benchmarking Enterprise Agents on Multi-Surface Knowledge Routing","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-01T01:35:09.639378Z"},"links":{"cited_paper":"/paper/2403.07718","citing_paper":"/paper/2607.25765"},"observation_digest":"sha256:0d92800c483bed3f9b568b00a95e8f5440f207eaed33c518412ca5b59bf38bc9","observation_id":"d47ac1dc-0e9c-4673-9b19-a1dea9104a1d","resolution":{"observed_at":"2026-08-01T01:35:09.639378Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.11005","last_updated":"2025-01-16T10:05:17Z","snapshot_observed_at":"2026-08-07T17:43:32.240365Z","submitted_at":"2024-06-25T20:23:15Z","title":"RAGBench: Explainable Benchmark for Retrieval-Augmented Generation Systems","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.11005","snapshot_observed_at":"2026-08-01T01:35:09.644671Z","title":"Ragbench: Explainable benchmark for retrieval-augmented generation systems.arXiv preprint arXiv:2407.11005, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.25765","last_updated":"2026-07-28T14:19:59Z","snapshot_observed_at":"2026-08-09T00:29:07.819944Z","submitted_at":"2026-07-28T14:19:59Z","title":"WorkSurface-Bench: Benchmarking Enterprise Agents on Multi-Surface Knowledge Routing","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-01T01:35:09.644671Z"},"links":{"cited_paper":"/paper/2407.11005","citing_paper":"/paper/2607.25765"},"observation_digest":"sha256:85c3cfcc2af16e7504e1407d5dfdd447825824ca32dffba4b12601703fcdd751","observation_id":"23be38f8-5a13-4e70-a8ed-f26f92066377","resolution":{"observed_at":"2026-08-01T01:35:09.644671Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T01:35:09.649427Z","title":"Datasheets for datasets.Communications of the ACM, 64(12):86–92, 2021","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2607.25765","last_updated":"2026-07-28T14:19:59Z","snapshot_observed_at":"2026-08-09T00:29:07.819944Z","submitted_at":"2026-07-28T14:19:59Z","title":"WorkSurface-Bench: Benchmarking Enterprise Agents on Multi-Surface Knowledge Routing","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-01T01:35:09.649427Z"},"links":{"citing_paper":"/paper/2607.25765"},"observation_digest":"sha256:8a71bbff952edd8eb918b27816c957c0e8fe12f2d5ba6353a6dcef6eb76cb2dc","observation_id":"c37a0e75-f368-4552-bee4-22a2d8bfda77","resolution":{"observed_at":"2026-08-01T01:35:09.649427Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T01:35:09.653588Z","title":"Gemini 3.1 Pro Preview","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2607.25765","last_updated":"2026-07-28T14:19:59Z","snapshot_observed_at":"2026-08-09T00:29:07.819944Z","submitted_at":"2026-07-28T14:19:59Z","title":"WorkSurface-Bench: Benchmarking Enterprise Agents on Multi-Surface Knowledge Routing","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-01T01:35:09.653588Z"},"links":{"citing_paper":"/paper/2607.25765"},"observation_digest":"sha256:ee014f9627dfc2a1402f251ebacff244853cc84d22b1c58a965398acf6fe554c","observation_id":"70bc1e83-f463-49e4-9499-62e0d886ccd4","resolution":{"observed_at":"2026-08-01T01:35:09.653588Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.03128","last_updated":"2024-12-04T19:49:02Z","snapshot_observed_at":"2026-07-06T16:27:55.509482Z","submitted_at":"2023-10-04T19:39:26Z","title":"MetaTool Benchmark for Large Language Models: Deciding Whether to Use Tools and Which to Use","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.03128","snapshot_observed_at":"2026-08-01T01:35:09.657698Z","title":"Metatool benchmark for large language models: Deciding whether to use tools and which to use","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.25765","last_updated":"2026-07-28T14:19:59Z","snapshot_observed_at":"2026-08-09T00:29:07.819944Z","submitted_at":"2026-07-28T14:19:59Z","title":"WorkSurface-Bench: Benchmarking Enterprise Agents on Multi-Surface Knowledge Routing","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-01T01:35:09.657698Z"},"links":{"cited_paper":"/paper/2310.03128","citing_paper":"/paper/2607.25765"},"observation_digest":"sha256:35fb5841ae7eb1ce90b2519b6e6cb22a2ec9fe556ab925850a275b5aaeec07a9","observation_id":"cf78261e-76a9-4938-8b66-59b2e68a263f","resolution":{"observed_at":"2026-08-01T01:35:09.657698Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T01:35:09.662068Z","title":"T2-ragbench: Text-and-table benchmark for evaluating retrieval-augmented generation","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2607.25765","last_updated":"2026-07-28T14:19:59Z","snapshot_observed_at":"2026-08-09T00:29:07.819944Z","submitted_at":"2026-07-28T14:19:59Z","title":"WorkSurface-Bench: Benchmarking Enterprise Agents on Multi-Surface Knowledge Routing","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-01T01:35:09.662068Z"},"links":{"citing_paper":"/paper/2607.25765"},"observation_digest":"sha256:6c80b031e7469868e9c33437461cf369bc6d35fc1bf2203a48bcd0e45d3491d3","observation_id":"d573cf1f-e909-4551-a683-deed9e74dc31","resolution":{"observed_at":"2026-08-01T01:35:09.662068Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T01:35:09.666217Z","title":"Can llm already serve as a database interface? a big bench for large-scale database grounded text-to-sqls","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.25765","last_updated":"2026-07-28T14:19:59Z","snapshot_observed_at":"2026-08-09T00:29:07.819944Z","submitted_at":"2026-07-28T14:19:59Z","title":"WorkSurface-Bench: Benchmarking Enterprise Agents on Multi-Surface Knowledge Routing","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-01T01:35:09.666217Z"},"links":{"citing_paper":"/paper/2607.25765"},"observation_digest":"sha256:d8bedbdf52cd6aa6aab04910395452a577636c8a666f02bb4f8fd4d12da31d67","observation_id":"13e9d000-48aa-4d85-9293-652f686b6874","resolution":{"observed_at":"2026-08-01T01:35:09.666217Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.08244","last_updated":"2023-10-25T06:54:12Z","snapshot_observed_at":"2026-08-02T00:07:12.855748Z","submitted_at":"2023-04-14T14:05:32Z","title":"API-Bank: A Comprehensive Benchmark for Tool-Augmented LLMs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.08244","snapshot_observed_at":"2026-08-01T01:35:09.670186Z","title":"Api-bank: A comprehensive benchmark for tool-augmented llms","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.25765","last_updated":"2026-07-28T14:19:59Z","snapshot_observed_at":"2026-08-09T00:29:07.819944Z","submitted_at":"2026-07-28T14:19:59Z","title":"WorkSurface-Bench: Benchmarking Enterprise Agents on Multi-Surface Knowledge Routing","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-01T01:35:09.670186Z"},"links":{"cited_paper":"/paper/2304.08244","citing_paper":"/paper/2607.25765"},"observation_digest":"sha256:515519b75e6d068bb0db5e0c88f5bcbb3cb53f576cce543cbb3cfa7f44403bad","observation_id":"f53554aa-2c8a-4243-a619-d863f7379586","resolution":{"observed_at":"2026-08-01T01:35:09.670186Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.13178","last_updated":"2024-12-23T20:12:48Z","snapshot_observed_at":"2026-07-06T17:19:44.130417Z","submitted_at":"2024-01-24T01:51:00Z","title":"AgentBoard: An Analytical Evaluation Board of Multi-turn LLM Agents","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.13178","snapshot_observed_at":"2026-08-01T01:35:09.674463Z","title":"Agentboard: An analytical evaluation board of multi-turn llm agents","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.25765","last_updated":"2026-07-28T14:19:59Z","snapshot_observed_at":"2026-08-09T00:29:07.819944Z","submitted_at":"2026-07-28T14:19:59Z","title":"WorkSurface-Bench: Benchmarking Enterprise Agents on Multi-Surface Knowledge Routing","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-01T01:35:09.674463Z"},"links":{"cited_paper":"/paper/2401.13178","citing_paper":"/paper/2607.25765"},"observation_digest":"sha256:82d4042459c807c88ab66a8eee9d4f71b0e09ec7d02f6677780f1715020ce3eb","observation_id":"b53841d8-7de1-4d38-8c00-2d2f42968e7d","resolution":{"observed_at":"2026-08-01T01:35:09.674463Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T01:35:09.678676Z","title":"GPT-4o mini: Advancing cost-efficient intelligence","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.25765","last_updated":"2026-07-28T14:19:59Z","snapshot_observed_at":"2026-08-09T00:29:07.819944Z","submitted_at":"2026-07-28T14:19:59Z","title":"WorkSurface-Bench: Benchmarking Enterprise Agents on Multi-Surface Knowledge Routing","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-01T01:35:09.678676Z"},"links":{"citing_paper":"/paper/2607.25765"},"observation_digest":"sha256:02d6e6f8e62d863aecb6d7c7b786a331efbbbda49354e1ea69a622943042f78b","observation_id":"168d70a1-82e9-49e3-b54a-fedeaf692a12","resolution":{"observed_at":"2026-08-01T01:35:09.678676Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T01:35:09.682692Z","title":"GPT-5.5 System Card","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2607.25765","last_updated":"2026-07-28T14:19:59Z","snapshot_observed_at":"2026-08-09T00:29:07.819944Z","submitted_at":"2026-07-28T14:19:59Z","title":"WorkSurface-Bench: Benchmarking Enterprise Agents on Multi-Surface Knowledge Routing","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-01T01:35:09.682692Z"},"links":{"citing_paper":"/paper/2607.25765"},"observation_digest":"sha256:f8f993535d97e40cf91e13d7a5e1a7e2ebd0876859800d63095ffc1d81d49a5c","observation_id":"050dc6bf-3376-48a6-9e59-72c21fa67835","resolution":{"observed_at":"2026-08-01T01:35:09.682692Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.16789","last_updated":"2023-10-03T14:45:48Z","snapshot_observed_at":"2026-07-06T16:00:46.542753Z","submitted_at":"2023-07-31T15:56:53Z","title":"ToolLLM: Facilitating Large Language Models to Master 16000+ Real-world APIs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.16789","snapshot_observed_at":"2026-08-01T01:35:09.686595Z","title":"Toolllm: Facilitating large language models to master 16000+ real-world apis","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.25765","last_updated":"2026-07-28T14:19:59Z","snapshot_observed_at":"2026-08-09T00:29:07.819944Z","submitted_at":"2026-07-28T14:19:59Z","title":"WorkSurface-Bench: Benchmarking Enterprise Agents on Multi-Surface Knowledge Routing","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-01T01:35:09.686595Z"},"links":{"cited_paper":"/paper/2307.16789","citing_paper":"/paper/2607.25765"},"observation_digest":"sha256:3e9f5b7b3d726340d17c6e154f82d5e184f16684eab1979a4b8a8372461cf408","observation_id":"937ca211-6651-424c-9c9f-62b5e0a5761d","resolution":{"observed_at":"2026-08-01T01:35:09.686595Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T01:35:09.690895Z","title":"Multimodalqa: Complex question answering over text, tables and images","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2607.25765","last_updated":"2026-07-28T14:19:59Z","snapshot_observed_at":"2026-08-09T00:29:07.819944Z","submitted_at":"2026-07-28T14:19:59Z","title":"WorkSurface-Bench: Benchmarking Enterprise Agents on Multi-Surface Knowledge Routing","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-01T01:35:09.690895Z"},"links":{"citing_paper":"/paper/2607.25765"},"observation_digest":"sha256:9a45eae1fc1284095a8b09dadcd51b0535708a243dc82be1df261510783dac8a","observation_id":"69b6dc52-38aa-42c8-b03b-6318710af74b","resolution":{"observed_at":"2026-08-01T01:35:09.690895Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T01:35:09.695237Z","title":"Workspace-bench 1.0: Benchmarking ai agents on workspace tasks with large-scale file dependencies, 2026","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2607.25765","last_updated":"2026-07-28T14:19:59Z","snapshot_observed_at":"2026-08-09T00:29:07.819944Z","submitted_at":"2026-07-28T14:19:59Z","title":"WorkSurface-Bench: Benchmarking Enterprise Agents on Multi-Surface Knowledge Routing","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-01T01:35:09.695237Z"},"links":{"citing_paper":"/paper/2607.25765"},"observation_digest":"sha256:72d778906a7e9f49d0bab49f964a6ce4e2fd1ec9e359614fd57b9934f2af629c","observation_id":"9c56d0a6-22b9-4bda-a314-acb627dc4b35","resolution":{"observed_at":"2026-08-01T01:35:09.695237Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.18901","last_updated":"2024-07-26T17:55:45Z","snapshot_observed_at":"2026-08-01T20:19:05.798018Z","submitted_at":"2024-07-26T17:55:45Z","title":"AppWorld: A Controllable World of Apps and People for Benchmarking Interactive Coding Agents","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.18901","snapshot_observed_at":"2026-08-01T01:35:09.699367Z","title":"Appworld: A controllable world of apps and people for benchmarking interactive coding agents.arXiv preprint arXiv:2407.18901, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.25765","last_updated":"2026-07-28T14:19:59Z","snapshot_observed_at":"2026-08-09T00:29:07.819944Z","submitted_at":"2026-07-28T14:19:59Z","title":"WorkSurface-Bench: Benchmarking Enterprise Agents on Multi-Surface Knowledge Routing","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-01T01:35:09.699367Z"},"links":{"cited_paper":"/paper/2407.18901","citing_paper":"/paper/2607.25765"},"observation_digest":"sha256:0653b69174fea5f934850f7dfcc2d169d604c38048a62402a7f9910689da99b4","observation_id":"d44a1db0-31dc-4bd9-b67d-6af4d440d3ac","resolution":{"observed_at":"2026-08-01T01:35:09.699367Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2404.13207","last_updated":"2024-10-20T18:59:02Z","snapshot_observed_at":"2026-08-09T19:51:07.032799Z","submitted_at":"2024-04-19T22:54:54Z","title":"STaRK: Benchmarking LLM Retrieval on Textual and Relational Knowledge Bases","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.13207","snapshot_observed_at":"2026-08-01T01:35:09.703423Z","title":"Stark: Benchmarking llm retrieval on semi-structured knowledge bases","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.25765","last_updated":"2026-07-28T14:19:59Z","snapshot_observed_at":"2026-08-09T00:29:07.819944Z","submitted_at":"2026-07-28T14:19:59Z","title":"WorkSurface-Bench: Benchmarking Enterprise Agents on Multi-Surface Knowledge Routing","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-01T01:35:09.703423Z"},"links":{"cited_paper":"/paper/2404.13207","citing_paper":"/paper/2607.25765"},"observation_digest":"sha256:659e260cb97f385dd8e41746bc8694ab6ec6c739f771e3fab319352fa15e5dba","observation_id":"8d5f06d9-8e64-4a82-b815-d4c5e75ca588","resolution":{"observed_at":"2026-08-01T01:35:09.703423Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2505.11180","last_updated":"2025-05-16T12:31:29Z","snapshot_observed_at":"2026-08-07T15:44:41.884510Z","submitted_at":"2025-05-16T12:31:29Z","title":"mmRAG: A Modular Benchmark for Retrieval-Augmented Generation over Text, Tables, and Knowledge Graphs","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.11180","snapshot_observed_at":"2026-08-01T01:35:09.707565Z","title":"mmRAG: A modular benchmark for retrieval- augmented generation over text, tables, and knowledge graphs.arXiv preprint arXiv:2505.11180, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.25765","last_updated":"2026-07-28T14:19:59Z","snapshot_observed_at":"2026-08-09T00:29:07.819944Z","submitted_at":"2026-07-28T14:19:59Z","title":"WorkSurface-Bench: Benchmarking Enterprise Agents on Multi-Surface Knowledge Routing","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-01T01:35:09.707565Z"},"links":{"cited_paper":"/paper/2505.11180","citing_paper":"/paper/2607.25765"},"observation_digest":"sha256:f616a352fd6edcfdc1df399ab377095f4d3b3cd9e7ff5c713fd3f4d2c007d88b","observation_id":"1c5d1a62-e920-485c-8767-bb8fffed445a","resolution":{"observed_at":"2026-08-01T01:35:09.707565Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2412.14161","last_updated":"2025-09-10T08:35:19Z","snapshot_observed_at":"2026-08-01T16:27:28.241667Z","submitted_at":"2024-12-18T18:55:40Z","title":"TheAgentCompany: Benchmarking LLM Agents on Consequential Real World Tasks","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2412.14161","snapshot_observed_at":"2026-08-01T01:35:09.711939Z","title":"Theagentcompany: Benchmarking llm agents on consequential real world tasks.arXiv preprint arXiv:2412.14161, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.25765","last_updated":"2026-07-28T14:19:59Z","snapshot_observed_at":"2026-08-09T00:29:07.819944Z","submitted_at":"2026-07-28T14:19:59Z","title":"WorkSurface-Bench: Benchmarking Enterprise Agents on Multi-Surface Knowledge Routing","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-01T01:35:09.711939Z"},"links":{"cited_paper":"/paper/2412.14161","citing_paper":"/paper/2607.25765"},"observation_digest":"sha256:1be84071229ebaefb51cf29bf41f8838fec401d0480bfac0c2a2c6c5bb7d9bda","observation_id":"03f12314-471f-4369-98d8-f8f3e7a87277","resolution":{"observed_at":"2026-08-01T01:35:09.711939Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.12045","last_updated":"2024-06-17T19:33:08Z","snapshot_observed_at":"2026-08-08T21:08:39.676079Z","submitted_at":"2024-06-17T19:33:08Z","title":"$\\tau$-bench: A Benchmark for Tool-Agent-User Interaction in Real-World Domains","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.12045","snapshot_observed_at":"2026-08-01T01:35:09.716221Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.25765","last_updated":"2026-07-28T14:19:59Z","snapshot_observed_at":"2026-08-09T00:29:07.819944Z","submitted_at":"2026-07-28T14:19:59Z","title":"WorkSurface-Bench: Benchmarking Enterprise Agents on Multi-Surface Knowledge Routing","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-01T01:35:09.716221Z"},"links":{"cited_paper":"/paper/2406.12045","citing_paper":"/paper/2607.25765"},"observation_digest":"sha256:cdf4ba396892bff72f6d3bed71a8f27ce207f0a9879859b4570947b53d34f18b","observation_id":"2c1f1068-9342-4d9e-9f30-a8dc84c7acc4","resolution":{"observed_at":"2026-08-01T01:35:09.716221Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-01T01:35:09.720250Z","title":"For workspace task 107: what is the total sales amount across the Asia Pacific re- gion?","venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2607.25765","last_updated":"2026-07-28T14:19:59Z","snapshot_observed_at":"2026-08-09T00:29:07.819944Z","submitted_at":"2026-07-28T14:19:59Z","title":"WorkSurface-Bench: Benchmarking Enterprise Agents on Multi-Surface Knowledge Routing","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-01T01:35:09.720250Z"},"links":{"citing_paper":"/paper/2607.25765"},"observation_digest":"sha256:26be9a6f96c6a26b1a2d7074f5d9957f193169cada1338ba3306b4687e443367","observation_id":"c1bce01c-2193-483b-9abc-13f622f9cd71","resolution":{"observed_at":"2026-08-01T01:35:09.720250Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2607.25765","last_updated":"2026-07-28T14:19:59Z","latest_version":1,"primary_category":"cs.CL","snapshot_observed_at":"2026-08-09T00:29:07.819944Z","submitted_at":"2026-07-28T14:19:59Z","title":"WorkSurface-Bench: Benchmarking Enterprise Agents on Multi-Surface Knowledge Routing"},"reference_resolution":{"displayed":25,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":25,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":25},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-09T06:31:02.800959+00:00","source":"crossref"},{"observed_at":"2026-08-09T06:30:57.326959+00:00","source":"retraction_watch"}],"thesis":"As of 10 August 2026, this Paper Citation Record lists 25 of 25 outbound references and 0 inbound Pith citation observations for arXiv:2607.25765."}