{"as_of":"2026-08-18T01:54:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:b98f3efb1dde95dcf0a67c658ff53947fe577c0c0e5a98251914761a42b92918","coverage":[{"denominator":97,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":97,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-15T16:09:59.965344Z","state":"measured"},{"denominator":102,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":102,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-17T06:30:58.91139+00:00","state":"measured"},{"denominator":5,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":5,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-04T05:55:44.614190Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":1,"source":"arxiv_reference","source_observed_at":"2026-08-05T02:28:24.338817Z","state":"measured"}],"external_citation_measurements":[{"count":0,"observed_at":"2026-08-05T02:28:24.338817Z","source":"arxiv_reference"}],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"cited_work":{"arxiv_id":"2509.09734","doi":"10.48550/arxiv.2509.09734","metadata_source":"arxiv_reference","pith_arxiv_id":"2509.09734","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mcp-agentbench: Evaluating real-world language agent performance with mcp-mediated tools","venue":"ArXiv.org","work_id":"03f7fc10-4b16-4fa7-8e0e-3a08b4133bc2","year":2025},"citing_paper":{"arxiv_id":"2602.10139","last_updated":"2026-04-26T01:34:23Z","snapshot_observed_at":"2026-08-14T10:04:04.434798Z","submitted_at":"2026-02-08T15:50:04Z","title":"Anonymization-Enhanced Privacy Protection for Mobile GUI Agents: Available but Invisible","version":3},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-05-16T05:59:32.583411Z"},"links":{"cited_paper":"/paper/2509.09734","citing_paper":"/paper/2602.10139"},"observation_digest":"sha256:791b8cea69637c342e900581f4c535b7e659cd91dcff5159a793375a941c2ab1","observation_id":"46516960-3898-4c92-b35a-c2d1002de80c","resolution":{"observed_at":"2026-05-16T06:00:40.667290Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2509.09734","snapshot_observed_at":"2026-08-04T05:55:44.614190Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2603.05637","last_updated":"2026-08-03T15:19:02Z","snapshot_observed_at":"2026-08-07T17:01:33.563160Z","submitted_at":"2026-03-05T19:47:26Z","title":"Real Faults in Model Context Protocol (MCP) Software: a Comprehensive Taxonomy","version":2},"reference_index":67,"source":"pdf_text","source_observed_at":"2026-08-04T05:55:44.614190Z"},"links":{"cited_paper":"/paper/2509.09734","citing_paper":"/paper/2603.05637"},"observation_digest":"sha256:b2eddc8ef09e29f44d8783f99cb7f6f164b969d0a5dd34e75681544275b87b1e","observation_id":"56e8193a-3f1c-40d5-9304-72274f6d06f6","resolution":{"observed_at":"2026-08-04T05:55:44.614190Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"cited_work":{"arxiv_id":"2509.09734","doi":"10.48550/arxiv.2509.09734","metadata_source":"arxiv_reference","pith_arxiv_id":"2509.09734","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mcp-agentbench: Evaluating real-world language agent performance with mcp-mediated tools","venue":"ArXiv.org","work_id":"03f7fc10-4b16-4fa7-8e0e-3a08b4133bc2","year":2025},"citing_paper":{"arxiv_id":"2604.03976","last_updated":"2026-05-04T20:58:24Z","snapshot_observed_at":"2026-07-06T22:53:01.835843Z","submitted_at":"2026-04-05T05:42:20Z","title":"Quantifying Trust: Financial Risk Management for Trustworthy AI Agents","version":2},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-05-13T17:16:17.464937Z"},"links":{"cited_paper":"/paper/2509.09734","citing_paper":"/paper/2604.03976"},"observation_digest":"sha256:b347f5e429f1b277f70588cca78956c9858e09ba8b92f62d1b4ea46577c90fa5","observation_id":"76a02012-7f93-47c7-853e-a62f37ecebef","resolution":{"observed_at":"2026-05-13T17:16:34.273960Z","resolver_source":"orphan_title_repair","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"cited_work":{"arxiv_id":"2509.09734","doi":"10.48550/arxiv.2509.09734","metadata_source":"arxiv_reference","pith_arxiv_id":"2509.09734","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mcp-agentbench: Evaluating real-world language agent performance with mcp-mediated tools","venue":"ArXiv.org","work_id":"03f7fc10-4b16-4fa7-8e0e-3a08b4133bc2","year":2025},"citing_paper":{"arxiv_id":"2604.07551","last_updated":"2026-04-08T19:53:26Z","snapshot_observed_at":"2026-08-15T04:36:06.031902Z","submitted_at":"2026-04-08T19:53:26Z","title":"MCP-DPT: A Defense-Placement Taxonomy and Coverage Analysis for Model Context Protocol Security","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-05-10T17:10:50.283791Z"},"links":{"cited_paper":"/paper/2509.09734","citing_paper":"/paper/2604.07551"},"observation_digest":"sha256:2bb873949fce95cf28f4f91af23056402beb5aae53781151fa11605be5024985","observation_id":"8143449b-8b9b-4f30-a53f-ec024858b5a8","resolution":{"observed_at":"2026-05-10T21:10:46.718657Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"cited_work":{"arxiv_id":"2509.09734","doi":"10.48550/arxiv.2509.09734","metadata_source":"arxiv_reference","pith_arxiv_id":"2509.09734","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mcp-agentbench: Evaluating real-world language agent performance with mcp-mediated tools","venue":"ArXiv.org","work_id":"03f7fc10-4b16-4fa7-8e0e-3a08b4133bc2","year":2025},"citing_paper":{"arxiv_id":"2605.14312","last_updated":"2026-05-14T03:23:51Z","snapshot_observed_at":"2026-08-02T14:41:28.516568Z","submitted_at":"2026-05-14T03:23:51Z","title":"Making OpenAPI Documentation Agent-Ready: Detecting Documentation and REST Smells with a Multi-Agent LLM System","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-05-15T02:44:01.980796Z"},"links":{"cited_paper":"/paper/2509.09734","citing_paper":"/paper/2605.14312"},"observation_digest":"sha256:ac68a9de3844155435ff331bf4a9c944d050ff8a0d3e311dc3c7e73e0e3a6fb6","observation_id":"f53f3936-eac1-4b42-9000-d6c08880459e","resolution":{"observed_at":"2026-05-15T02:48:33.931507Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2509.09734/citation-record","integrity":"/paper/2509.09734/integrity","json":"/paper/2509.09734/citation-record.json","paper":"/paper/2509.09734"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2303.08774","last_updated":"2024-03-04T06:01:33Z","snapshot_observed_at":"2026-08-17T09:58:46.058102Z","submitted_at":"2023-03-15T17:15:04Z","title":"GPT-4 Technical Report","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.08774","snapshot_observed_at":"2026-08-15T16:09:59.575503Z","title":"Gpt-4 technical report.arXiv preprint arXiv:2303.08774, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.575503Z"},"links":{"cited_paper":"/paper/2303.08774","citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:cec926c079acd2a6cc14c6ff0228c44d0f4cc2453903021e634ba5df337ac31b","observation_id":"f04dbfa6-bd77-4fa5-a37a-4a79c0534a81","resolution":{"observed_at":"2026-08-15T16:09:59.575503Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:09:59.579805Z","title":"Introducing computer use, a new claude 3.5 sonnet, and claude 3.5 haiku","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.579805Z"},"links":{"citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:27f790ec238349aa51b388600ef9a81f456636d71af494cca54a6022a91ab85d","observation_id":"1c176bf0-9594-424c-85a3-f87a194761ba","resolution":{"observed_at":"2026-08-15T16:09:59.579805Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:09:59.583800Z","title":"Claude 3.7 sonnet anthropic.https://www.anthropic.com/claude/sonnet, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.583800Z"},"links":{"citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:06cb26b3c16b2c817ad7a47af074b4a64b7fb27d43a4cb5dcacee711fbb7b049","observation_id":"d86f00c7-2f53-4b4f-98f1-81525ec2864f","resolution":{"observed_at":"2026-08-15T16:09:59.583800Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:09:59.588628Z","title":"Introducing the model context protocol anthropic","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.588628Z"},"links":{"citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:f90b714e325bb7a4300713a3b5b5dadeb9ec71d0fc5070d61dd778d87ea6d91b","observation_id":"13fd310f-916e-492d-8eb7-c9e0726846d5","resolution":{"observed_at":"2026-08-15T16:09:59.588628Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:09:59.592747Z","title":"chatmcp/mcprouter: api router for mcp servers.https://github.com/chatmcp/mcprouter, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.592747Z"},"links":{"citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:fae8b5609c72531a816bb02371c41b03490afa5ed54529485de0f54a6ed1ca4a","observation_id":"168e0b27-54d9-4bf3-9214-81fc55a441e2","resolution":{"observed_at":"2026-08-15T16:09:59.592747Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2507.06261","last_updated":"2025-12-19T14:25:46Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-07-07T17:36:04Z","title":"Gemini 2.5: Pushing the Frontier with Advanced Reasoning, Multimodality, Long Context, and Next Generation Agentic Capabilities","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2507.06261","snapshot_observed_at":"2026-08-15T16:09:59.597405Z","title":"Gemini 2.5: Pushing the frontier with advanced reasoning, multimodality, long context, and next generation agentic capabilities.arXiv preprint arXiv:2507.06261, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.597405Z"},"links":{"cited_paper":"/paper/2507.06261","citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:e30a0d9b41d098428f66d848df5fd636caaa0987d5f93d73cbf1bbfffd433cff","observation_id":"879c22a6-6d85-44f4-94b4-f24bbb89cca9","resolution":{"observed_at":"2026-08-15T16:09:59.597405Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:09:59.601414Z","title":"Deepseek-v3 technical report, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.601414Z"},"links":{"citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:0e0b4cb4670200cbae3e2b6b667ec56b800322c301b51d01a9dcce61e577796f","observation_id":"0d707125-de91-4dd9-ab94-c8e7357ff80c","resolution":{"observed_at":"2026-08-15T16:09:59.601414Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:09:59.605512Z","title":"Deepseek-r1: Incentivizing reasoning capability in llms via reinforcement learning, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.605512Z"},"links":{"citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:9cfb1b85ed8f5f2cfdef7056212dcfc97f506c7334e8ee751915111b4fb0d2f4","observation_id":"a69e02be-1b26-40b3-a1a4-625b84f0f6d9","resolution":{"observed_at":"2026-08-15T16:09:59.605512Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:09:59.608363Z","title":"Mcp-radar: A multi-dimensional benchmark for evaluating tool use capabilities in large language models.arXiv preprint arXiv:2505.16700, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.608363Z"},"links":{"citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:3beae6f108836d063edb2eb2cbb00878c7c9377389585532a81c607f4cddb4ce","observation_id":"4868df1c-a4c8-4185-9104-61bdfeb4cd00","resolution":{"observed_at":"2026-08-15T16:09:59.608363Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.01680","last_updated":"2024-04-19T01:15:16Z","snapshot_observed_at":"2026-08-10T13:10:07.804621Z","submitted_at":"2024-01-21T23:36:14Z","title":"Large Language Model based Multi-Agents: A Survey of Progress and Challenges","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.01680","snapshot_observed_at":"2026-08-15T16:09:59.612031Z","title":"Large language model based multi-agents: A survey of progress and challenges.arXiv preprint arXiv:2402.01680, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.612031Z"},"links":{"cited_paper":"/paper/2402.01680","citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:923bd9086f237606d3f19cf66ee369d4a57b2fce3e8d283fa6aaa6f5c2d7388b","observation_id":"47ac5a24-933a-4641-a266-90082df9a8ae","resolution":{"observed_at":"2026-08-15T16:09:59.612031Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:09:59.615822Z","title":"MetaGPT: Meta programming for a multi-agent collaborative framework","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.615822Z"},"links":{"citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:57e6ce891c4f8369896d89af21f704f4eec766c47df7feae3cd12389ced76550","observation_id":"b7f07f99-ff07-4187-b9aa-fb238c678cfe","resolution":{"observed_at":"2026-08-15T16:09:59.615822Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2503.23278","last_updated":"2025-10-07T07:13:32Z","snapshot_observed_at":"2026-08-14T03:34:08.418318Z","submitted_at":"2025-03-30T01:58:22Z","title":"Model Context Protocol (MCP): Landscape, Security Threats, and Future Research Directions","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.23278","snapshot_observed_at":"2026-08-15T16:09:59.619700Z","title":"Model context protocol (mcp): Landscape, security threats, and future research directions.arXiv preprint arXiv:2503.23278, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.619700Z"},"links":{"cited_paper":"/paper/2503.23278","citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:5a603226ea858d396eddc0e0e91316b340cd92cc97567071b3726d5a9eac891b","observation_id":"b6663164-3117-4521-a806-277fa4e005d3","resolution":{"observed_at":"2026-08-15T16:09:59.619700Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2212.10403","last_updated":"2023-05-26T17:59:33Z","snapshot_observed_at":"2026-08-06T19:23:30.079167Z","submitted_at":"2022-12-20T16:29:03Z","title":"Towards Reasoning in Large Language Models: A Survey","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2212.10403","snapshot_observed_at":"2026-08-15T16:09:59.623636Z","title":"Towards reasoning in large language models: A survey.arXiv preprint arXiv:2212.10403, 2022","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.623636Z"},"links":{"cited_paper":"/paper/2212.10403","citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:5d1382aee78782c167bfe3a96e8e4b94fb464ce74cda5941781dc29598a08db5","observation_id":"06a0485c-5b00-4639-8864-5c4b1bd3625a","resolution":{"observed_at":"2026-08-15T16:09:59.623636Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.02716","last_updated":"2024-02-05T04:25:24Z","snapshot_observed_at":"2026-08-15T04:53:46.192738Z","submitted_at":"2024-02-05T04:25:24Z","title":"Understanding the planning of LLM agents: A survey","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.02716","snapshot_observed_at":"2026-08-15T16:09:59.627932Z","title":"Understanding the planning of llm agents: A survey.arXiv preprint arXiv:2402.02716, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.627932Z"},"links":{"cited_paper":"/paper/2402.02716","citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:cc53d619a09ff1e578012b896b08c4f269091782fc089a847f7007e7ff90d888","observation_id":"07521cd2-d296-4137-989a-a8f50c0fc1f5","resolution":{"observed_at":"2026-08-15T16:09:59.627932Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:09:59.631917Z","title":"Api-bank: A comprehensive benchmark for tool-augmented llms, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.631917Z"},"links":{"citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:8c6d0041fd732975b05dafc1cd6a86ba715d407fb442644aec34185f76f6f705","observation_id":"6bcab3ce-d232-43fd-9c9f-8e146d990f11","resolution":{"observed_at":"2026-08-15T16:09:59.631917Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2409.02977","last_updated":"2025-12-03T03:33:35Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-09-04T15:59:41Z","title":"Large Language Model-Based Agents for Software Engineering: A Survey","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2409.02977","snapshot_observed_at":"2026-08-15T16:09:59.635780Z","title":"Large language model-based agents for software engineering: A survey.arXiv preprint arXiv:2409.02977, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.635780Z"},"links":{"cited_paper":"/paper/2409.02977","citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:e0e1c6988a92dc9990acab5a887b57ea9ab7781701ed12e3b86bff03fb7f3fb4","observation_id":"d062aab7-dc66-429e-9918-2665bfa91a1c","resolution":{"observed_at":"2026-08-15T16:09:59.635780Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2507.12806","last_updated":"2025-08-01T22:37:16Z","snapshot_observed_at":"2026-08-09T01:25:07.357969Z","submitted_at":"2025-07-17T05:46:27Z","title":"MCPEval: Automatic MCP-based Deep Evaluation for AI Agent Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2507.12806","snapshot_observed_at":"2026-08-15T16:09:59.639299Z","title":"Mcpeval: Automatic mcp-based deep evaluation for ai agent models.arXiv preprint arXiv:2507.12806, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.639299Z"},"links":{"cited_paper":"/paper/2507.12806","citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:e2812420998ff4d7d1387ca64e0682a35829bac3affc56745141520ac9c3209d","observation_id":"b57c8d05-524a-4962-90ca-703cb0946f1a","resolution":{"observed_at":"2026-08-15T16:09:59.639299Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2504.11094","last_updated":"2025-04-18T10:39:23Z","snapshot_observed_at":"2026-08-16T12:40:56.136470Z","submitted_at":"2025-04-15T11:40:12Z","title":"Evaluation Report on MCP Servers","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2504.11094","snapshot_observed_at":"2026-08-15T16:09:59.643148Z","title":"Evaluation report on mcp servers.arXiv preprint arXiv:2504.11094, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.643148Z"},"links":{"cited_paper":"/paper/2504.11094","citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:b1bb6419c5b4c48ef34cfca40a4f6a22fcd9c9a2b7211059cd5be40e344f1903","observation_id":"0e22d4aa-a2ec-4a60-b40b-c55e0199e61d","resolution":{"observed_at":"2026-08-15T16:09:59.643148Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.07945","last_updated":"2024-02-09T02:33:45Z","snapshot_observed_at":"2026-08-16T14:20:01.793685Z","submitted_at":"2024-02-09T02:33:45Z","title":"ScreenAgent: A Vision Language Model-driven Computer Control Agent","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.07945","snapshot_observed_at":"2026-08-15T16:09:59.648057Z","title":"Screenagent: A vision language model-driven computer control agent.arXiv preprint arXiv:2402.07945, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.648057Z"},"links":{"cited_paper":"/paper/2402.07945","citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:2aa93dbd37e153440521a0b58fae52f921b810a072380dd1462473f2f6a4023a","observation_id":"0e4953a0-bd75-46eb-bedd-8f58f5056c7e","resolution":{"observed_at":"2026-08-15T16:09:59.648057Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:09:59.652027Z","title":"Hello gpt-4o | openai.https://openai.com/index/hello-gpt-4o/, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.652027Z"},"links":{"citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:48feb8d4206edb6a2a55246765f1107f08f99cafe96e514f9b1e6e5829336373","observation_id":"f1533033-293f-4d9e-b5e3-81f3a4df788f","resolution":{"observed_at":"2026-08-15T16:09:59.652027Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:09:59.655974Z","title":"Openai o3-mini | openai.https://openai.com/index/openai-o3-mini/, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.655974Z"},"links":{"citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:7bbbacbab094c97ba825e41a651de931e312bb1f1f77dffbaf7f98ace8613c72","observation_id":"7a2132bd-0960-4387-99d3-732c6b30046a","resolution":{"observed_at":"2026-08-15T16:09:59.655974Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:09:59.659667Z","title":"Llm rankings | openrouter.https://openrouter.ai/rankings?view=month, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.659667Z"},"links":{"citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:1872d2122fcf028a7a9b661a73f29784c20090c3bad061c3060e3186233b9e53","observation_id":"205ea629-1552-464c-be80-0b4bf5ab1ad1","resolution":{"observed_at":"2026-08-15T16:09:59.659667Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.12373","last_updated":"2024-07-16T06:19:21Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-06-18T07:58:33Z","title":"WebCanvas: Benchmarking Web Agents in Online Environments","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.12373","snapshot_observed_at":"2026-08-15T16:09:59.663790Z","title":"Webcanvas: Benchmarking web agents in online environments.arXiv preprint arXiv:2406.12373, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.663790Z"},"links":{"cited_paper":"/paper/2406.12373","citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:a3ad4d9496fa336e0c2e4b0c67f09935f5ee3c599494cd3f9926be44e1aeb3ba","observation_id":"b23ac9c7-101b-4720-9d94-62a6f025607f","resolution":{"observed_at":"2026-08-15T16:09:59.663790Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:09:59.668957Z","title":"Gorilla: Large language model connected with massive apis.Advances in Neural Information Processing Systems, 37:126544–126565, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.668957Z"},"links":{"citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:2a048117986daf2ca3955a7fa6049f98bcf3704dd442ed5e62a7fe7800a79fca","observation_id":"cf9b569d-ad23-46f3-ba78-3c0be6bb8eee","resolution":{"observed_at":"2026-08-15T16:09:59.668957Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:09:59.671862Z","title":"Toolllm: Facilitating large language models to master 16000+ real-world apis, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.671862Z"},"links":{"citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:21c60679b3cc8e042e32e2f361fce52333655bc93bcf26106db8dc86d9e5e5ba","observation_id":"b6f77a3c-da88-4678-9a86-5ab38b87d8c1","resolution":{"observed_at":"2026-08-15T16:09:59.671862Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:09:59.674685Z","title":"Language agents: Foundations, prospects, and risks","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.674685Z"},"links":{"citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:d1a03131ab48ab3f22df519e4d0e8811ee4158fa0b7a3cb1b918c1c429a5d3d2","observation_id":"8fb6bc00-b3df-4228-93fc-aed0701492c5","resolution":{"observed_at":"2026-08-15T16:09:59.674685Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:09:59.678632Z","title":"Cognitive architectures for language agents.Transactions on Machine Learning Research, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.678632Z"},"links":{"citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:08010c8448c8f99870eccd9d6c73c4bb940ce5f3be5e537cfe142f4be3e1cfa9","observation_id":"07f89f4e-6fc2-4cdb-a061-2977d0d97121","resolution":{"observed_at":"2026-08-15T16:09:59.678632Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2507.20534","last_updated":"2026-02-03T04:57:00Z","snapshot_observed_at":"2026-08-16T14:37:33.548231Z","submitted_at":"2025-07-28T05:35:43Z","title":"Kimi K2: Open Agentic Intelligence","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2507.20534","snapshot_observed_at":"2026-08-15T16:09:59.682902Z","title":"Kimi k2: Open agentic intelligence.arXiv preprint arXiv:2507.20534, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.682902Z"},"links":{"cited_paper":"/paper/2507.20534","citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:48c7b59a65589f704284f4e5d27c0d60ad8af7c4d25c4162659f92c30692b5e8","observation_id":"c28884ec-3641-494b-b103-79fdd299ada3","resolution":{"observed_at":"2026-08-15T16:09:59.682902Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:09:59.687387Z","title":"Qwen3 technical report, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.687387Z"},"links":{"citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:5a8a1113c60bf61ac41148339152d14c5a71c1ca03c22de41859c18f4d7594d0","observation_id":"6fc6353a-0c45-4e43-9e29-6a556f706414","resolution":{"observed_at":"2026-08-15T16:09:59.687387Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:09:59.690716Z","title":"A survey on large language model based autonomous agents.Frontiers of Computer Science, 18(6):186345, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.690716Z"},"links":{"citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:499211cecdfe3e90849d47962a2db8b7cbbfcc89d147388185e66acfe61c80e4","observation_id":"f96705e4-3d83-40c3-877e-48ac8aba597f","resolution":{"observed_at":"2026-08-15T16:09:59.690716Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:09:59.694872Z","title":"Autogen: Enabling next-gen llm applications via multi-agent conversation, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.694872Z"},"links":{"citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:5b96d16b12b3f94f5fc8c486e7099d6c38b51806d9b41aeb7de0265c757ef859","observation_id":"14b090d7-c4c3-426b-8210-f1e21ef8ae1c","resolution":{"observed_at":"2026-08-15T16:09:59.694872Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:09:59.698092Z","title":"Osworld: Benchmarking multimodal agents for open-ended tasks in real computer environments.Advances in Neural Information Processing Systems, 37:52040–52094, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.698092Z"},"links":{"citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:ea2f3aa29cb66da80dcf75495358d7ae8a2aca70208cfe41ba19ba6123a76fde","observation_id":"044d3413-1474-4962-abc7-413accfe3377","resolution":{"observed_at":"2026-08-15T16:09:59.698092Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:09:59.701522Z","title":"An illusion of progress? assessing the current state of web agents.arXiv preprint arXiv:2504.01382, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.701522Z"},"links":{"citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:9eebc5c383a5c28090c4267ec6a0136fc1f55075d6d40e099d1af20588c86edb","observation_id":"30ad9d7a-68b4-4ef4-9ef4-1fcb6118cc0d","resolution":{"observed_at":"2026-08-15T16:09:59.701522Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:10:00.861579Z","title":"Patil, Ion Stoica, and Joseph E","venue":null,"work_id":"80c68fce-0a7d-4679-bd70-28dc412bd766","year":2024},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.705683Z"},"links":{"citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:02c546a64d954f51bba1827032fce9dd433de412b71bd60be6891db77142c927","observation_id":"47498be3-a365-4029-9a3e-7b47fdecd2a5","resolution":{"observed_at":"2026-08-15T16:10:00.865496Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:09:59.709630Z","title":"Swe-agent: Agent-computer interfaces enable automated software engineering.Advances in Neural Information Processing Systems, 37:50528–50652, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.709630Z"},"links":{"citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:89e1b56608069632d46ed347c2e20e5ff0c4b51763acc6f2539c652f82229122","observation_id":"62bb34c7-8896-46fb-b289-a2fea6208e5c","resolution":{"observed_at":"2026-08-15T16:09:59.709630Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.12045","last_updated":"2024-06-17T19:33:08Z","snapshot_observed_at":"2026-08-17T20:31:29.818313Z","submitted_at":"2024-06-17T19:33:08Z","title":"$\\tau$-bench: A Benchmark for Tool-Agent-User Interaction in Real-World Domains","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.12045","snapshot_observed_at":"2026-08-15T16:09:59.714886Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.714886Z"},"links":{"cited_paper":"/paper/2406.12045","citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:46c76e63c5588db7548b8314f38761d408be4c4aff606c73f5eed5f51983cd29","observation_id":"f8d13c9c-1275-4834-8797-61d36ac0433e","resolution":{"observed_at":"2026-08-15T16:09:59.714886Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.18279","last_updated":"2025-05-06T15:08:00Z","snapshot_observed_at":"2026-08-17T10:40:45.253009Z","submitted_at":"2024-11-27T12:13:39Z","title":"Large Language Model-Brained GUI Agents: A Survey","version":12},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.18279","snapshot_observed_at":"2026-08-15T16:09:59.718887Z","title":"Large language model-brained gui agents: A survey.arXiv preprint arXiv:2411.18279, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.718887Z"},"links":{"cited_paper":"/paper/2411.18279","citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:2bdaec544dd6c7bb5064c68da2b82a34cfe10b8f1598279275ee0275ec9c8376","observation_id":"c0642870-3162-40f6-973f-62484b96e688","resolution":{"observed_at":"2026-08-15T16:09:59.718887Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:09:59.722052Z","title":"Judging llm-as-a-judge with mt-bench and chatbot arena.Advances in Neural Information Processing Systems, 36:46595–46623, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.722052Z"},"links":{"citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:4fcee31cb66286bac52f7a0623a46bde9cdcdd5b507110ab9ae5ff5c0b006f3b","observation_id":"37da5206-ca5d-4fd5-a2cd-18bf17786bef","resolution":{"observed_at":"2026-08-15T16:09:59.722052Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:10:00.840798Z","title":"Complexfuncbench: Exploring multi-step and constrained function calling under long-context scenario, 2025","venue":null,"work_id":"7ba60250-b51b-454a-842c-166089967ff9","year":2025},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.725222Z"},"links":{"citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:64410ba5818b58d9045a1add008ae2799adc00bcef11b55791df5c1915f03105","observation_id":"d4982fd5-e71f-4645-9f2b-453ae62a5819","resolution":{"observed_at":"2026-08-15T16:10:00.844547Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.13854","last_updated":"2024-04-16T15:13:18Z","snapshot_observed_at":"2026-08-14T11:14:55.351653Z","submitted_at":"2023-07-25T22:59:32Z","title":"WebArena: A Realistic Web Environment for Building Autonomous Agents","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.13854","snapshot_observed_at":"2026-08-15T16:09:59.729245Z","title":"low-pass-rate","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.729245Z"},"links":{"cited_paper":"/paper/2307.13854","citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:2368311948296c30705f5f2901d7ca31bbcfa390bbd4b0afe65a4fa7e0a496b4","observation_id":"063cfff0-3a05-4f1d-8618-a3f802bab087","resolution":{"observed_at":"2026-08-15T16:09:59.729245Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:10:00.832060Z","title":null,"venue":null,"work_id":"7fb5c920-564b-4e8a-b0b3-352aec192fad","year":null},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.732808Z"},"links":{"citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:1d71d55a40fe5e6864fb7faf7488b450281bee9a94c895af9dfe0e5b25a9f9e4","observation_id":"5d63803d-fe75-4be9-b0bc-702cebfdbc14","resolution":{"observed_at":"2026-08-15T16:10:00.834958Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:10:00.811925Z","title":null,"venue":null,"work_id":"59ea35a6-e557-4dc3-89f0-20982b54b84c","year":null},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.739668Z"},"links":{"citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:cbe11f42b43b131c7e9a981fe6357e440043c7dc74629312837d2d5c99576bdb","observation_id":"50ead5b2-c716-43c4-8504-8bd1f0e6f188","resolution":{"observed_at":"2026-08-15T16:10:00.815240Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:10:00.801778Z","title":"this weekend","venue":null,"work_id":"51bba25e-3706-40f9-b9a5-279648adfcbc","year":2025},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.743339Z"},"links":{"citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:36d34aadc122e224bb45cfc26501f3d58f7812c3e09f50a1b8a1d56a1b5a970f","observation_id":"f5c7a093-9bbb-4acf-8f77-4f14248ce584","resolution":{"observed_at":"2026-08-15T16:10:00.804961Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:10:00.791507Z","title":null,"venue":null,"work_id":"b4ed306b-f315-4148-a8ae-a61528f224cf","year":null},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.746948Z"},"links":{"citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:3742e96efcfdd2c3c8ca6749429572debbd26140dfd5720d4c061cd87235d1d3","observation_id":"0872d06d-f242-441f-8d5a-7492a3e83ad5","resolution":{"observed_at":"2026-08-15T16:10:00.795014Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:10:00.781756Z","title":null,"venue":null,"work_id":"40b9e521-53b4-441b-bdae-88bb5951eb1d","year":null},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.750875Z"},"links":{"citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:94b6b75abb6efb77628993842414b905ba4ff10c7bc11075940845f7feda657e","observation_id":"5f60a7dc-3ab1-485f-9421-1160cac72b58","resolution":{"observed_at":"2026-08-15T16:10:00.785292Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:10:00.772230Z","title":null,"venue":null,"work_id":"2721b942-9d49-4e8c-a197-b05c54d8dcee","year":null},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.755547Z"},"links":{"citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:353d1432e9dcd4529ad804d9dab86f826ef3caa9f1d9e2fb482729df189fa35b","observation_id":"417c3555-9d27-4221-a08c-a629b8caff02","resolution":{"observed_at":"2026-08-15T16:10:00.775369Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:10:00.761679Z","title":null,"venue":null,"work_id":"166154bf-47cb-4836-8437-5de65c52cae2","year":null},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.759952Z"},"links":{"citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:8284db3002160dd60aa8b857bbf8111adcfabb89b47c0c5eab2fe1c8e6a56ce4","observation_id":"d1b3d1ca-c724-4e2c-8680-6c976401d345","resolution":{"observed_at":"2026-08-15T16:10:00.765669Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:10:00.751236Z","title":null,"venue":null,"work_id":"ff0b530e-71a8-45e6-b7fc-143d69a2f008","year":null},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.764247Z"},"links":{"citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:accdb54e02e451502941dd7e707a8628c29c7d77ff2abf7fabf84db40863ece3","observation_id":"55b850f5-ea9a-430d-aa06-ca4b086e4429","resolution":{"observed_at":"2026-08-15T16:10:00.755150Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:10:00.740456Z","title":null,"venue":null,"work_id":"79a444d0-4327-4c0a-82c6-d8ddee7b6ca3","year":null},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.768616Z"},"links":{"citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:c6073b15062e60574649a6d5ffa232b62504ced6b9f2299946dcdc394fe4c1a8","observation_id":"58542e5c-d3b3-4f89-98b9-788b22e529fe","resolution":{"observed_at":"2026-08-15T16:10:00.744955Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:10:00.731833Z","title":null,"venue":null,"work_id":"68283b6e-1f0d-4f36-a890-d6ffb033caa2","year":null},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.771695Z"},"links":{"citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:43b23b77844bbbc0e690c7667ab38d742ffbfeb597cf9d7052b4c52c1240636c","observation_id":"4240e96b-e641-428f-b72a-0cb0fe4d9d21","resolution":{"observed_at":"2026-08-15T16:10:00.734739Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:10:00.717676Z","title":null,"venue":null,"work_id":"c95c642d-b363-4dd2-a3ef-d0a8decbce6b","year":null},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.775130Z"},"links":{"citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:fdb1c7d7c52cfbce333bf8c45f910206c431b75d76469a55657007639d559934","observation_id":"89bf7a33-d3d8-40a8-b2bb-40449aab706a","resolution":{"observed_at":"2026-08-15T16:10:00.725461Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:10:00.707593Z","title":null,"venue":null,"work_id":"815f3a59-f21c-4896-81bd-2527e6c0accc","year":null},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.779420Z"},"links":{"citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:7d5246d3de64fc8c5232e8f603bc30731b704b3aa581fbdb3499dc274d7bac17","observation_id":"b77ee098-db42-4074-a3f0-327b8376719e","resolution":{"observed_at":"2026-08-15T16:10:00.710547Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:10:00.696882Z","title":null,"venue":null,"work_id":"257f43dd-8cfc-4104-a7da-a4f38a5e292e","year":null},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.783120Z"},"links":{"citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:7636168cdc9d8502f275497b02515b17f4b09ea086c5510e6a1e4ae762a133fb","observation_id":"ece40590-03c0-47a4-b3fa-432c743b8615","resolution":{"observed_at":"2026-08-15T16:10:00.700614Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:10:00.687440Z","title":null,"venue":null,"work_id":"2bdf91c2-e68b-41a2-831d-43a239dfeafd","year":null},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.787685Z"},"links":{"citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:0cde0621b9a5524fe98bd60af3c9b7847df6572eb24dd331332ccf08b4f614b7","observation_id":"a6c449b7-9933-47c4-a29b-d0b96ba0c29b","resolution":{"observed_at":"2026-08-15T16:10:00.690565Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:10:00.674606Z","title":null,"venue":null,"work_id":"af7a6b01-2611-48de-af52-d6f77849f7be","year":null},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.792288Z"},"links":{"citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:0be085d33d5371ad69b5890d80159d2232e396aa4f69504cdc4b1af4094692da","observation_id":"5c7c3dc5-a122-4fb3-af84-35aec7076c99","resolution":{"observed_at":"2026-08-15T16:10:00.678315Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:10:00.665240Z","title":"Contextual Component Generation System Prompt You are a Realistic MCP Server Tool Scenario Designer","venue":null,"work_id":"f1b5cdab-5c89-4b7b-8556-764fd1a8f461","year":null},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.796437Z"},"links":{"citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:945f17794ccb5dde055edf95eb8bdb07be77a17a845d2e36f85fd9762e51abc6","observation_id":"7e5d1bab-289c-4e38-b10b-0e3077876a7e","resolution":{"observed_at":"2026-08-15T16:10:00.668751Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:10:00.655068Z","title":"Explain why these tools are necessary and sufficient","venue":null,"work_id":"f455bf77-65f9-42c9-843f-4ae8187aef0f","year":null},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.799863Z"},"links":{"citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:2ddfbfb9180482eb2223e771430bf993e282271d92c1426de35db40de4a1ad5e","observation_id":"c697fc60-be5d-42fd-a47d-018c3d214872","resolution":{"observed_at":"2026-08-15T16:10:00.658691Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:10:00.645997Z","title":"</user_profile>","venue":null,"work_id":"77760619-e97e-40c1-b1ef-98b51f8a8ba4","year":null},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.803218Z"},"links":{"citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:5a1bbe8c743edbcb9307e11c743b10b50409360b8e7e49e7fcde773f1a7f94b8","observation_id":"abd66cc0-994c-4349-b839-d97ee7fe5178","resolution":{"observed_at":"2026-08-15T16:10:00.648996Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:10:00.636771Z","title":"</scenario>","venue":null,"work_id":"5558de6b-87d7-4d0b-9b4c-f7ec1ca6ada8","year":null},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.806551Z"},"links":{"citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:5e5a87ca1a95999821fdc5a0039fc9129cb3fb6de4e311b26cbe169588b09401","observation_id":"ed6153d1-d3f3-4303-b612-f4f45a947cca","resolution":{"observed_at":"2026-08-15T16:10:00.639882Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:10:00.626685Z","title":"</objective> Parameter Sourcing Requirements All tool parameters must come from:","venue":null,"work_id":"c900ca1a-df7a-4592-9550-2934a9bafe3f","year":null},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.809854Z"},"links":{"citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:11dff8546e31d93407cdbc04019d68182ecaf366903c340e52675fc27de58d0a","observation_id":"985203c1-be12-4404-bb49-fe827d4242b4","resolution":{"observed_at":"2026-08-15T16:10:00.629853Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:10:00.616581Z","title":null,"venue":null,"work_id":"45ff5167-2b98-4c30-8c7d-544d18cd0d68","year":null},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.814043Z"},"links":{"citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:436ba6ce6e84fbdb16ca6f4de1528a1cac86670fd9c73f971e8fc13e3b0c8e0d","observation_id":"da2982c8-d296-49f4-83c4-816f00e39d0b","resolution":{"observed_at":"2026-08-15T16:10:00.619600Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:10:00.823114Z","title":null,"venue":null,"work_id":"189248b7-c927-4d48-b388-c942b765c7f2","year":null},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.818146Z"},"links":{"citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:d625fe7a1a5d61c23db359f41cbafea607c1bd53626719fb88407f867c0dbe3b","observation_id":"b707530b-fdb7-4c48-aea1-bbde2172ee01","resolution":{"observed_at":"2026-08-15T16:10:00.826281Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:10:00.605043Z","title":"All necessary information must be available in the initial request or derived from tool usage","venue":null,"work_id":"a6d0448c-b192-4a9e-a9f6-260a6f3f487c","year":null},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":64,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.821334Z"},"links":{"citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:4657ee4314f654782ac84a8a98b2423c2ec85c87efad3111b9c15023c7127c3d","observation_id":"6516e0c3-8ae9-4065-b601-88e630b6ceaa","resolution":{"observed_at":"2026-08-15T16:10:00.608870Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:10:00.593307Z","title":null,"venue":null,"work_id":"f3c011b1-e3ef-41ff-acea-acc6912bc05b","year":null},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":65,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.824797Z"},"links":{"citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:ec17cc753d51e649416a109f6f41a611a30e04d9ebd0258d9ade79bb786291ea","observation_id":"2557775e-36cc-4a04-97d8-784974b4c29b","resolution":{"observed_at":"2026-08-15T16:10:00.597445Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:10:00.583231Z","title":null,"venue":null,"work_id":"bd9d536d-c75b-4767-b337-fda8109e774a","year":null},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":66,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.828077Z"},"links":{"citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:6d544327ee152eaf5ca934b63f66cc64b8bb2a0b7b7eec65cfa2e32b2a600c47","observation_id":"1bb2157d-dfaf-464c-90c0-50b8e6f87b0d","resolution":{"observed_at":"2026-08-15T16:10:00.586246Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:10:00.573368Z","title":null,"venue":null,"work_id":"c25d077d-52b3-4513-99cd-84f4bb9da995","year":null},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":67,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.830910Z"},"links":{"citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:a9e3ec278a4465dac4fa85cfe9d84d5abc2da5fa5a9654c1116441c0fcbfbd9d","observation_id":"657a37d3-6f32-4958-8e1a-3ee772e0cf2f","resolution":{"observed_at":"2026-08-15T16:10:00.577060Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:10:00.563083Z","title":"18 ReAcT Assistant Prompt You are an advanced AI assistant with access to Model Context Protocol (MCP) servers","venue":null,"work_id":"f36fb59d-6fac-4d95-b2cc-93827dd9843f","year":null},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":68,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.834090Z"},"links":{"citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:9cccd48114de95c61c7c5f1466b2c3892c0aa4d92f18ec061724e7a0317e72bc","observation_id":"83c73c7c-e788-4558-a8d5-ef4a9ad8a69f","resolution":{"observed_at":"2026-08-15T16:10:00.566663Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:10:00.529267Z","title":null,"venue":null,"work_id":"0f8c66e5-66d4-4322-9bf0-7b6bd06df0ea","year":null},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":71,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.844340Z"},"links":{"citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:bcc6820b466361927833c2f67dc61814698b211e4726a4c35b055a75cbf323cd","observation_id":"229d5c6e-0714-427d-949d-7cb44f0453a1","resolution":{"observed_at":"2026-08-15T16:10:00.532847Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:10:00.453336Z","title":null,"venue":null,"work_id":"ff1af438-ad47-4eec-bb3a-1b9585e83f26","year":null},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":78,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.870421Z"},"links":{"citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:faa413163045d832fa8ab734fb365cf1d873b48008b6caf8c988f64851ed5f44","observation_id":"1d356728-faec-46e2-8d6a-2ee761e6afa7","resolution":{"observed_at":"2026-08-15T16:10:00.457398Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:10:00.431126Z","title":"Use tools strategically but don’t overcomplicate simple requests that can be answered directly","venue":null,"work_id":"32a02202-ee2e-4844-bf3e-3c129fe82f97","year":null},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":80,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.876280Z"},"links":{"citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:6a00082104174b5666df6f613b5d37c2060e42c25daebbbb9f28207668c75fe7","observation_id":"5cd8c327-a81c-4e8d-b2be-4d1b440e7ff7","resolution":{"observed_at":"2026-08-15T16:10:00.435982Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:10:00.421742Z","title":null,"venue":null,"work_id":"c78d6fba-40e6-4d91-8368-182278d38251","year":null},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":81,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.878985Z"},"links":{"citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:788c5bdf9cbf321a222603040ea9aa57cd6bb9e0c5a793c02391b04fe88cf145","observation_id":"4043924b-b56e-4208-b72c-077fad4a032e","resolution":{"observed_at":"2026-08-15T16:10:00.425198Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:10:00.411120Z","title":null,"venue":null,"work_id":"9912e2e6-fb3b-4a12-89f9-bc7f25704cdc","year":null},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":82,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.881951Z"},"links":{"citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:d07fe8a3bcd149c16542736be39e0ed799072b6e80227de8a113c60ce29d1a36","observation_id":"e1829e18-510d-4851-bdaa-c7c0d8b168f0","resolution":{"observed_at":"2026-08-15T16:10:00.414636Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:10:00.400977Z","title":null,"venue":null,"work_id":"74f134f4-5a77-4a06-b267-bf596ef10433","year":null},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":83,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.884910Z"},"links":{"citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:8ebd1bd7d12902dcbd450b7f20e0ed5b0e03dbc4a5b121983330642086c258e2","observation_id":"a83a8bf4-61af-4858-b53f-3c8fee1578d6","resolution":{"observed_at":"2026-08-15T16:10:00.404351Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:10:00.391589Z","title":"calculate","venue":null,"work_id":"1b09f79f-bbd8-47c5-950d-3186119b8496","year":null},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":84,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.888074Z"},"links":{"citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:6a5c138483a01a8c5d874c5259ef91d1efac3e785d511823acaccf6fb2b97c05","observation_id":"a1663766-b26c-43c2-bbc6-329752d2741d","resolution":{"observed_at":"2026-08-15T16:10:00.395335Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:10:00.382066Z","title":null,"venue":null,"work_id":"7f1b88ae-ec75-4bfe-bc2c-ec050acbf913","year":null},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":85,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.891084Z"},"links":{"citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:5b206db3c3b5822fcca55c129982a99744d06ba76dcfd1149fdac23b5c040946","observation_id":"a89b75c7-6712-4809-867a-fe0623828ba6","resolution":{"observed_at":"2026-08-15T16:10:00.385190Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:10:00.372095Z","title":"name\": \"selected_tool_name","venue":null,"work_id":"7f3c1b8f-46c2-4d37-8257-571279dee0e0","year":null},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":86,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.894774Z"},"links":{"citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:a4f49d0c0a170fd4eb67579d501a4d7feed2b42f4e31d320937c6f392f8e1296","observation_id":"70dc9e5c-e84f-4938-8385-896eb8c5b574","resolution":{"observed_at":"2026-08-15T16:10:00.375172Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:10:00.550302Z","title":null,"venue":null,"work_id":"a131ef5a-07c5-4110-ad92-fc2a649af4ee","year":null},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":87,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.898845Z"},"links":{"citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:ccee82d05658fedc94ae6e55f3dd8955b24b779d896af277e0630fe0805baf56","observation_id":"6b993c42-c56e-4c77-a97a-ccdac8625a57","resolution":{"observed_at":"2026-08-15T16:10:00.554806Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:10:00.540178Z","title":null,"venue":null,"work_id":"3aeaa6a7-3126-4918-b55e-476dd6477af8","year":null},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":88,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.901865Z"},"links":{"citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:65c38c84b6449f93e47e396c146777555757975e45461dc72bb7756e59346421","observation_id":"97ab0760-1241-4af9-9c9f-26a9a314982d","resolution":{"observed_at":"2026-08-15T16:10:00.544081Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:10:00.362524Z","title":null,"venue":null,"work_id":"c55969a7-58c5-4f79-b8c5-cb0a43024e24","year":null},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":89,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.905292Z"},"links":{"citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:ce35622cac17329ac134b88ea8edd6002d9e2bea4abaa87f991097f8f8565776","observation_id":"e0fb5e69-d851-460a-aa9a-b6ecb96aa77f","resolution":{"observed_at":"2026-08-15T16:10:00.365791Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:10:00.519103Z","title":null,"venue":null,"work_id":"c9b04d0e-0677-4be1-b49d-a78d4a556306","year":null},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":90,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.908836Z"},"links":{"citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:41604c243914f6f97f844a3330c0fcace95043ace3773b3311e16b8bd70976a9","observation_id":"c30e4b07-e76a-45ff-9b89-249153480e2a","resolution":{"observed_at":"2026-08-15T16:10:00.522858Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:10:00.506985Z","title":null,"venue":null,"work_id":"f6b5a472-8c12-439d-b5d5-0194231d4f33","year":null},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":91,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.911505Z"},"links":{"citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:78830ce128f939d5af59b0e51bf4c05749edd07f2e6fbcdba0b08c3f3dc38c00","observation_id":"cf0e8efc-d0b2-4d7f-a7a2-c0daed721ae5","resolution":{"observed_at":"2026-08-15T16:10:00.510747Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:10:00.496309Z","title":null,"venue":null,"work_id":"25e4467d-03b7-46d1-aaaa-0a907277eab4","year":null},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":92,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.914609Z"},"links":{"citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:e1fe0d09a6ea65b1edd9cbfda69321d4534fcd219c6e3610d39872c3f0c4900f","observation_id":"10e81301-20ef-4921-8e66-420614433afc","resolution":{"observed_at":"2026-08-15T16:10:00.499240Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:10:00.487091Z","title":null,"venue":null,"work_id":"3671732d-89f1-45a4-a2d1-98b6243a5cb6","year":null},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":93,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.917497Z"},"links":{"citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:c1b7993d2ec979609dbf9acb3b94c2d7afb073c37350a00d0200dc2823792e0c","observation_id":"ebd9e01b-b608-48c8-8200-8ca908dad167","resolution":{"observed_at":"2026-08-15T16:10:00.490476Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:10:00.477963Z","title":null,"venue":null,"work_id":"f053a554-54c9-43ef-9e11-4992a4c36a78","year":null},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":94,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.920809Z"},"links":{"citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:3070ae62a30dbe5f6c4f41a03c7263411fb726d199d5cd305b00647c3703fa3d","observation_id":"da615eed-1126-4ee7-bc81-ed4c88104657","resolution":{"observed_at":"2026-08-15T16:10:00.480993Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:10:00.466197Z","title":null,"venue":null,"work_id":"dcca3ce8-b978-498a-a522-26506b9d5d31","year":null},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":95,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.923510Z"},"links":{"citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:74d36e0705d610e29ae517de5b12dafd26e80a007cf5f7ad3a9a5343f5d89639","observation_id":"7acaa1f0-89ab-458c-be71-7ed9df5c4008","resolution":{"observed_at":"2026-08-15T16:10:00.470719Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:10:00.352505Z","title":null,"venue":null,"work_id":"b55700a3-f385-4706-b568-72758c67e256","year":null},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":96,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.926469Z"},"links":{"citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:ed45b2a347300bb82a21522aca382c14d503aea7e8a9b22799146d2b0997af3e","observation_id":"6c08c922-68d2-4611-8db4-4fa37716a2b6","resolution":{"observed_at":"2026-08-15T16:10:00.355565Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:10:00.443237Z","title":null,"venue":null,"work_id":"81e29e6c-ec49-4128-9bb4-03f7bb961b04","year":null},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":97,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.929765Z"},"links":{"citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:8fbcd1c74b101ae35b07bda0e35d41d25040e61d4076f29ddc0c896178dc6b6d","observation_id":"83bc4bf8-18b4-42ea-ba74-8b8cfc6e65fb","resolution":{"observed_at":"2026-08-15T16:10:00.447203Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:10:00.340784Z","title":"Use tools strategically but don’t overcomplicate simple requests that can be answered directly","venue":null,"work_id":"f9534cc9-00ca-4b8e-8a6a-5f79a49709fb","year":null},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":98,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.933367Z"},"links":{"citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:24f391540626c16ffeed5edf49c269db9628c4e759c62a16632894d9734964ab","observation_id":"22ba905f-b340-4c0d-9088-a5b4825eb36b","resolution":{"observed_at":"2026-08-15T16:10:00.344957Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:10:00.330088Z","title":"typically,","venue":null,"work_id":"1f7f6a28-71e9-423e-82f7-fb08581653fb","year":null},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":99,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.937178Z"},"links":{"citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:f58074da8bbbecef6cb2d465b68297a988fe1cf94725bbdda72eb542fe2fef82","observation_id":"2b55d9f9-b424-4265-a1ab-a285d7e48451","resolution":{"observed_at":"2026-08-15T16:10:00.333928Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:10:00.319479Z","title":"proof\" or","venue":null,"work_id":"81c2a349-6744-4fbb-b96a-43eb154fe7a8","year":null},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":100,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.940243Z"},"links":{"citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:d2bebd3f06d2b9fac583eeb2ce78622b7bc91603711843fdc2f4dc07e83143b9","observation_id":"29e93816-391c-42a4-8d64-4db7d28a3d90","resolution":{"observed_at":"2026-08-15T16:10:00.323241Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:10:00.308452Z","title":"Verification details","venue":null,"work_id":"6b240919-3dae-499a-bb60-566c7e198587","year":null},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":101,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.943340Z"},"links":{"citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:c8db76d2b823f9aeffa3b4d101b76ab9ac74d32207c9d310bff8b0172b60c018","observation_id":"fea0b363-9c63-4e45-877e-6b34cb413f67","resolution":{"observed_at":"2026-08-15T16:10:00.311853Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:10:00.296963Z","title":null,"venue":null,"work_id":"9058b07b-9952-4d35-a976-d7a7e53e99ed","year":null},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":102,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.947012Z"},"links":{"citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:4ee7750eb13ef2cc87b2172f1339ffe4850cd605aaea736e882f4772df102b34","observation_id":"5d597700-554b-4119-abdf-bab294d4c826","resolution":{"observed_at":"2026-08-15T16:10:00.300334Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:10:00.286219Z","title":null,"venue":null,"work_id":"d166641c-4d79-410d-b5d0-ec525cdf1026","year":null},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":103,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.950392Z"},"links":{"citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:31ce6557f137ef632404918db15fa0c8aff3e2ae5f31fd6810e6eb8396653adf","observation_id":"87ac6600-c55e-4873-ae41-d263e5e79900","resolution":{"observed_at":"2026-08-15T16:10:00.289757Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:10:00.275469Z","title":"Lacks verification details about data source","venue":null,"work_id":"0ecb60f4-81bf-48d3-b042-f8baecc649e3","year":null},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":104,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.954473Z"},"links":{"citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:24e3e5d3d6563ac9e32a8ac4d2c1dda3ce347853d86c857abcc880604f9eb7cd","observation_id":"096281c7-054f-47e6-94ae-0470684d2cd1","resolution":{"observed_at":"2026-08-15T16:10:00.278836Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:10:00.264723Z","title":"• Process explanations • Formatting differences 5.Decide: • External data + Core need met = PASS • Knowledge only = FAIL","venue":null,"work_id":"8b70ba01-5e2b-471a-8a86-53ee40ab66dc","year":null},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":105,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.957941Z"},"links":{"citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:2d5dd36b531d79fe9a4560cabc105e9f216488a10143a34e114a186446e960ab","observation_id":"77a03be7-8536-4173-93a7-86b25ce8b96e","resolution":{"observed_at":"2026-08-15T16:10:00.269044Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:10:00.251031Z","title":"Is there specific external data that helps the user?","venue":null,"work_id":"83d57746-86dc-46d7-b1d3-abd8978f56bb","year":null},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":106,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.961419Z"},"links":{"citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:d9bfe5b9457efaa0852c879d9f5dabcbcbf2c9af45345ac3f15c8f5d85ff4705","observation_id":"2e1e93ba-941f-4c01-b977-d593ba76da6e","resolution":{"observed_at":"2026-08-15T16:10:00.256273Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T16:10:00.239776Z","title":"nice-to-have","venue":null,"work_id":"f91c5079-dc4b-4e43-a323-123c018c6620","year":null},"citing_paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools","version":1},"reference_index":107,"source":"pdf_text","source_observed_at":"2026-08-15T16:09:59.965344Z"},"links":{"citing_paper":"/paper/2509.09734"},"observation_digest":"sha256:1dcc988c137114f196d5296481eca2f49a442049223b0645db734e24b3350b55","observation_id":"6f305a3b-9d9f-43e6-a629-a1a386f1f675","resolution":{"observed_at":"2026-08-15T16:10:00.243841Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2509.09734","last_updated":"2025-09-10T14:08:40Z","latest_version":1,"primary_category":"cs.CL","snapshot_observed_at":"2026-08-17T21:51:47.803898Z","submitted_at":"2025-09-10T14:08:40Z","title":"MCP-AgentBench: Evaluating Real-World Language Agent Performance with MCP-Mediated Tools"},"reference_resolution":{"displayed":97,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":76,"verified_exact":0,"verified_fuzzy":21},"total_outbound_references":97},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"thesis":"As of 18 August 2026, this Paper Citation Record lists 97 of 97 outbound references and 5 inbound Pith citation observations for arXiv:2509.09734."}