{"as_of":"2026-08-07T16:27:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:08fb074b602221261f3d140c597925f114c5d9ce6489b5d831a72b2554e5f85f","coverage":[{"denominator":23,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":23,"source":"paper_references, paper_reference_links","source_observed_at":"2026-05-15T20:51:41.198767Z","state":"measured"},{"denominator":104,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":104,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-07T06:34:17.273281+00:00","state":"measured"},{"denominator":81,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":81,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T14:12:00.366784Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":1,"source":"pith","source_observed_at":"2026-08-05T02:28:24.338817Z","state":"measured"}],"external_citation_measurements":[{"count":12,"observed_at":"2026-08-05T02:28:24.338817Z","source":"pith"}],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2304.08244","last_updated":"2023-10-25T06:54:12Z","snapshot_observed_at":"2026-08-02T00:07:12.855748Z","submitted_at":"2023-04-14T14:05:32Z","title":"API-Bank: A Comprehensive Benchmark for Tool-Augmented LLMs","version":2},"cited_work":{"arxiv_id":"2304.08244","doi":"10.48550/arxiv.2304.08244","metadata_source":"pith","pith_arxiv_id":"2304.08244","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"API-Bank: A Comprehensive Benchmark for Tool-Augmented LLMs","venue":"cs.CL","work_id":"a20d9332-ab34-485c-a060-1ba47cc98930","year":2023},"citing_paper":{"arxiv_id":"2305.17144","last_updated":"2023-06-01T09:18:01Z","snapshot_observed_at":"2026-07-06T15:34:05.442952Z","submitted_at":"2023-05-25T17:59:49Z","title":"Ghost in the Minecraft: Generally Capable Agents for Open-World Environments via Large Language Models with Text-based Knowledge and Memory","version":2},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-05-15T18:37:20.889942Z"},"links":{"cited_paper":"/paper/2304.08244","citing_paper":"/paper/2305.17144"},"observation_digest":"sha256:70e6e3e0fdab367dc97aff57e3d7e64fafb3607b9661de46c94409d450d37164","observation_id":"cea748f9-ea52-4557-aae7-92268217187b","resolution":{"observed_at":"2026-05-15T20:51:41.275571Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-07-14T18:20:23.914543+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-14T18:20:23.914543+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.08244","last_updated":"2023-10-25T06:54:12Z","snapshot_observed_at":"2026-08-02T00:07:12.855748Z","submitted_at":"2023-04-14T14:05:32Z","title":"API-Bank: A Comprehensive Benchmark for Tool-Augmented LLMs","version":2},"cited_work":{"arxiv_id":"2304.08244","doi":"10.48550/arxiv.2304.08244","metadata_source":"pith","pith_arxiv_id":"2304.08244","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"API-Bank: A Comprehensive Benchmark for Tool-Augmented LLMs","venue":"cs.CL","work_id":"a20d9332-ab34-485c-a060-1ba47cc98930","year":2023},"citing_paper":{"arxiv_id":"2306.05301","last_updated":"2023-09-07T12:20:45Z","snapshot_observed_at":"2026-07-06T15:40:20.267344Z","submitted_at":"2023-06-08T15:46:32Z","title":"ToolAlpaca: Generalized Tool Learning for Language Models with 3000 Simulated Cases","version":2},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-15T23:03:48.426204Z"},"links":{"cited_paper":"/paper/2304.08244","citing_paper":"/paper/2306.05301"},"observation_digest":"sha256:35a73ff3ddddf9500583c37225286eeec736816b9ba238ceb2d272b3a4be00c5","observation_id":"3254d231-6c20-4e95-8005-f8cd33db1e87","resolution":{"observed_at":"2026-05-15T23:03:48.465611Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-07-14T18:20:23.914543+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-14T18:20:23.914543+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.08244","last_updated":"2023-10-25T06:54:12Z","snapshot_observed_at":"2026-08-02T00:07:12.855748Z","submitted_at":"2023-04-14T14:05:32Z","title":"API-Bank: A Comprehensive Benchmark for Tool-Augmented LLMs","version":2},"cited_work":{"arxiv_id":"2304.08244","doi":"10.48550/arxiv.2304.08244","metadata_source":"pith","pith_arxiv_id":"2304.08244","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"API-Bank: A Comprehensive Benchmark for Tool-Augmented LLMs","venue":"cs.CL","work_id":"a20d9332-ab34-485c-a060-1ba47cc98930","year":2023},"citing_paper":{"arxiv_id":"2306.06070","last_updated":"2023-12-09T05:57:46Z","snapshot_observed_at":"2026-07-06T15:40:50.807376Z","submitted_at":"2023-06-09T17:44:31Z","title":"Mind2Web: Towards a Generalist Agent for the Web","version":3},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-05-15T20:05:15.992207Z"},"links":{"cited_paper":"/paper/2304.08244","citing_paper":"/paper/2306.06070"},"observation_digest":"sha256:e68b54b0ac1f28f0bd892cc60f7d1c1eebff0e448138b0c72bd6e7151a6206c7","observation_id":"afbfab5c-6e56-4b0c-a33a-bf472e40bc3c","resolution":{"observed_at":"2026-05-15T20:51:41.275571Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-07-14T18:20:23.914543+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-14T18:20:23.914543+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.08244","last_updated":"2023-10-25T06:54:12Z","snapshot_observed_at":"2026-08-02T00:07:12.855748Z","submitted_at":"2023-04-14T14:05:32Z","title":"API-Bank: A Comprehensive Benchmark for Tool-Augmented LLMs","version":2},"cited_work":{"arxiv_id":"2304.08244","doi":"10.48550/arxiv.2304.08244","metadata_source":"pith","pith_arxiv_id":"2304.08244","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"API-Bank: A Comprehensive Benchmark for Tool-Augmented LLMs","venue":"cs.CL","work_id":"a20d9332-ab34-485c-a060-1ba47cc98930","year":2023},"citing_paper":{"arxiv_id":"2308.11432","last_updated":"2025-03-02T04:04:03Z","snapshot_observed_at":"2026-08-02T07:33:16.089197Z","submitted_at":"2023-08-22T13:30:37Z","title":"A Survey on Large Language Model based Autonomous Agents","version":7},"reference_index":70,"source":"pdf_text","source_observed_at":"2026-05-15T04:03:00.340349Z"},"links":{"cited_paper":"/paper/2304.08244","citing_paper":"/paper/2308.11432"},"observation_digest":"sha256:b1ab40be375febfdc22de2bd325a0c1469857238bad08f3eb70f9dc8b0e1d294","observation_id":"e1c98ac3-f231-4fb0-a5f2-f65336e93c80","resolution":{"observed_at":"2026-05-15T20:51:41.275571Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-07-14T18:20:23.914543+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-14T18:20:23.914543+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.08244","last_updated":"2023-10-25T06:54:12Z","snapshot_observed_at":"2026-08-02T00:07:12.855748Z","submitted_at":"2023-04-14T14:05:32Z","title":"API-Bank: A Comprehensive Benchmark for Tool-Augmented LLMs","version":2},"cited_work":{"arxiv_id":"2304.08244","doi":"10.48550/arxiv.2304.08244","metadata_source":"pith","pith_arxiv_id":"2304.08244","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"API-Bank: A Comprehensive Benchmark for Tool-Augmented LLMs","venue":"cs.CL","work_id":"a20d9332-ab34-485c-a060-1ba47cc98930","year":2023},"citing_paper":{"arxiv_id":"2311.12983","last_updated":"2023-11-21T20:34:47Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-11-21T20:34:47Z","title":"GAIA: a benchmark for General AI Assistants","version":1},"reference_index":52,"source":"arxiv_source","source_observed_at":"2026-05-12T15:46:03.247029Z"},"links":{"cited_paper":"/paper/2304.08244","citing_paper":"/paper/2311.12983"},"observation_digest":"sha256:16579f5b73bccf41ae03a7ff67614b1cf39a2efd40bea8906bc1894da48fa38e","observation_id":"648ad77e-fc10-466a-8385-0b591ab0782c","resolution":{"observed_at":"2026-05-15T20:51:41.275571Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-07-14T18:20:23.914543+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-14T18:20:23.914543+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.08244","last_updated":"2023-10-25T06:54:12Z","snapshot_observed_at":"2026-08-02T00:07:12.855748Z","submitted_at":"2023-04-14T14:05:32Z","title":"API-Bank: A Comprehensive Benchmark for Tool-Augmented LLMs","version":2},"cited_work":{"arxiv_id":"2304.08244","doi":"10.48550/arxiv.2304.08244","metadata_source":"pith","pith_arxiv_id":"2304.08244","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"API-Bank: A Comprehensive Benchmark for Tool-Augmented LLMs","venue":"cs.CL","work_id":"a20d9332-ab34-485c-a060-1ba47cc98930","year":2023},"citing_paper":{"arxiv_id":"2409.00557","last_updated":"2026-04-29T05:49:57Z","snapshot_observed_at":"2026-07-06T19:08:44.269514Z","submitted_at":"2024-08-31T23:06:12Z","title":"Learning to Ask: When LLM Agents Meet Unclear Instruction","version":4},"reference_index":8,"source":"arxiv_source","source_observed_at":"2026-05-23T21:08:42.276002Z"},"links":{"cited_paper":"/paper/2304.08244","citing_paper":"/paper/2409.00557"},"observation_digest":"sha256:19b7db1d2113588471a4ad6b05a06afbbbc07378ab54be06731e151822e9dbdf","observation_id":"93c65495-1427-4f0a-b1cb-4e78ee6135a6","resolution":{"observed_at":"2026-05-23T21:13:28.115525Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-07-14T18:20:23.914543+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-14T18:20:23.914543+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.08244","last_updated":"2023-10-25T06:54:12Z","snapshot_observed_at":"2026-08-02T00:07:12.855748Z","submitted_at":"2023-04-14T14:05:32Z","title":"API-Bank: A Comprehensive Benchmark for Tool-Augmented LLMs","version":2},"cited_work":{"arxiv_id":"2304.08244","doi":"10.48550/arxiv.2304.08244","metadata_source":"pith","pith_arxiv_id":"2304.08244","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"API-Bank: A Comprehensive Benchmark for Tool-Augmented LLMs","venue":"cs.CL","work_id":"a20d9332-ab34-485c-a060-1ba47cc98930","year":2023},"citing_paper":{"arxiv_id":"2504.19793","last_updated":"2025-08-24T03:28:21Z","snapshot_observed_at":"2026-08-01T07:24:09.967062Z","submitted_at":"2025-04-28T13:36:43Z","title":"Prompt Injection Attack to Tool Selection in LLM Agents","version":3},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-05-16T17:08:28.933831Z"},"links":{"cited_paper":"/paper/2304.08244","citing_paper":"/paper/2504.19793"},"observation_digest":"sha256:c9f5dd0df1044acc76295aaf3398301dc434b4a91d50d31a8083c1a19262facb","observation_id":"165a7397-528a-4460-b29d-b9c72399118a","resolution":{"observed_at":"2026-05-16T17:08:29.053456Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-07-14T18:20:23.914543+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-14T18:20:23.914543+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.08244","last_updated":"2023-10-25T06:54:12Z","snapshot_observed_at":"2026-08-02T00:07:12.855748Z","submitted_at":"2023-04-14T14:05:32Z","title":"API-Bank: A Comprehensive Benchmark for Tool-Augmented LLMs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.08244","snapshot_observed_at":"2026-08-07T14:12:00.366784Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.19683","last_updated":"2025-05-26T08:44:53Z","snapshot_observed_at":"2026-08-07T14:06:07.358887Z","submitted_at":"2025-05-26T08:44:53Z","title":"Large Language Models for Planning: A Comprehensive and Systematic Survey","version":1},"reference_index":137,"source":"pdf_text","source_observed_at":"2026-08-07T14:12:00.366784Z"},"links":{"cited_paper":"/paper/2304.08244","citing_paper":"/paper/2505.19683"},"observation_digest":"sha256:90e493ca52fb0ed34299dd1885818074f763618b6dadec3ec0afb1da154b016a","observation_id":"34492b2d-85d1-4026-b19d-de76312de76c","resolution":{"observed_at":"2026-08-07T14:12:00.366784Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.08244","last_updated":"2023-10-25T06:54:12Z","snapshot_observed_at":"2026-08-02T00:07:12.855748Z","submitted_at":"2023-04-14T14:05:32Z","title":"API-Bank: A Comprehensive Benchmark for Tool-Augmented LLMs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.08244","snapshot_observed_at":"2026-08-07T13:53:18.595877Z","title":"Api-bank: A comprehensive benchmark for tool-augmented llms","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.21577","last_updated":"2025-08-25T13:40:36Z","snapshot_observed_at":"2026-08-07T13:42:38.045978Z","submitted_at":"2025-05-27T08:35:05Z","title":"RepoMaster: Autonomous Exploration and Understanding of GitHub Repositories for Complex Task Solving","version":3},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-07T13:53:18.595877Z"},"links":{"cited_paper":"/paper/2304.08244","citing_paper":"/paper/2505.21577"},"observation_digest":"sha256:ac4c5f267e8c29aab820b69519e3c8c353411c9b9456538f8d58ad9d11b41b4c","observation_id":"ba463c84-86c6-460f-8cd0-b4124e2350f5","resolution":{"observed_at":"2026-08-07T13:53:18.595877Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.08244","last_updated":"2023-10-25T06:54:12Z","snapshot_observed_at":"2026-08-02T00:07:12.855748Z","submitted_at":"2023-04-14T14:05:32Z","title":"API-Bank: A Comprehensive Benchmark for Tool-Augmented LLMs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.08244","snapshot_observed_at":"2026-08-07T13:21:42.186784Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.00042","last_updated":"2025-05-28T08:39:35Z","snapshot_observed_at":"2026-08-07T13:12:36.585457Z","submitted_at":"2025-05-28T08:39:35Z","title":"Enhancing Tool Learning in Large Language Models with Hierarchical Error Checklists","version":1},"reference_index":13,"source":"arxiv_source","source_observed_at":"2026-08-07T13:21:42.186784Z"},"links":{"cited_paper":"/paper/2304.08244","citing_paper":"/paper/2506.00042"},"observation_digest":"sha256:7ce2bfac3673e4bcee63e9b1ff2288ad950176ada47a1353cfc57a71bd0e07ad","observation_id":"a97d9b4c-bd01-40c3-9269-b843841da7f4","resolution":{"observed_at":"2026-08-07T13:21:42.186784Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.08244","last_updated":"2023-10-25T06:54:12Z","snapshot_observed_at":"2026-08-02T00:07:12.855748Z","submitted_at":"2023-04-14T14:05:32Z","title":"API-Bank: A Comprehensive Benchmark for Tool-Augmented LLMs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.08244","snapshot_observed_at":"2026-08-07T11:56:25.466965Z","title":"Api-bank: A comprehensive benchmark for tool-augmented llms.arXiv preprint arXiv:2304.08244, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.01056","last_updated":"2025-06-24T06:27:29Z","snapshot_observed_at":"2026-08-07T11:49:43.837276Z","submitted_at":"2025-06-01T15:48:53Z","title":"MCP-Zero: Active Tool Discovery for Autonomous LLM Agents","version":4},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-07T11:56:25.466965Z"},"links":{"cited_paper":"/paper/2304.08244","citing_paper":"/paper/2506.01056"},"observation_digest":"sha256:fb0604b9e4163405ef726b4a743fa0add4059f1bcc686d262a9d5e68f725131c","observation_id":"d9cd42b4-7d6f-46cc-afe1-52765e1357df","resolution":{"observed_at":"2026-08-07T11:56:25.466965Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.08244","last_updated":"2023-10-25T06:54:12Z","snapshot_observed_at":"2026-08-02T00:07:12.855748Z","submitted_at":"2023-04-14T14:05:32Z","title":"API-Bank: A Comprehensive Benchmark for Tool-Augmented LLMs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.08244","snapshot_observed_at":"2026-08-07T11:36:38.560529Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.01859","last_updated":"2025-06-02T16:48:11Z","snapshot_observed_at":"2026-08-07T11:30:07.185749Z","submitted_at":"2025-06-02T16:48:11Z","title":"CONFETTI: Conversational Function-Calling Evaluation Through Turn-Level Interactions","version":1},"reference_index":10,"source":"arxiv_source","source_observed_at":"2026-08-07T11:36:38.560529Z"},"links":{"cited_paper":"/paper/2304.08244","citing_paper":"/paper/2506.01859"},"observation_digest":"sha256:05b1fb492a27f6e9527f8097bfb99300a2bc35007a5101042707ee29658357c0","observation_id":"3e78bb37-d63d-4262-a402-f070637bde39","resolution":{"observed_at":"2026-08-07T11:36:38.560529Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.08244","last_updated":"2023-10-25T06:54:12Z","snapshot_observed_at":"2026-08-02T00:07:12.855748Z","submitted_at":"2023-04-14T14:05:32Z","title":"API-Bank: A Comprehensive Benchmark for Tool-Augmented LLMs","version":2},"cited_work":{"arxiv_id":"2304.08244","doi":"10.48550/arxiv.2304.08244","metadata_source":"pith","pith_arxiv_id":"2304.08244","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"API-Bank: A Comprehensive Benchmark for Tool-Augmented LLMs","venue":"cs.CL","work_id":"a20d9332-ab34-485c-a060-1ba47cc98930","year":2023},"citing_paper":{"arxiv_id":"2506.19500","last_updated":"2025-10-31T14:24:22Z","snapshot_observed_at":"2026-07-06T21:46:53.087123Z","submitted_at":"2025-06-24T10:39:07Z","title":"NaviAgent: Bilevel Planning on Tool Navigation Graph for Large-Scale Orchestration","version":2},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-05-19T08:12:10.542542Z"},"links":{"cited_paper":"/paper/2304.08244","citing_paper":"/paper/2506.19500"},"observation_digest":"sha256:6a4d21a717af9c6cdd16983ef0c606d6f9988cce9c6cd2f559a08eca372bfd2e","observation_id":"dbee5958-f76d-4e3a-ae8c-9eb264c3da18","resolution":{"observed_at":"2026-05-19T08:13:01.834735Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-07-14T18:20:23.914543+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-14T18:20:23.914543+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.08244","last_updated":"2023-10-25T06:54:12Z","snapshot_observed_at":"2026-08-02T00:07:12.855748Z","submitted_at":"2023-04-14T14:05:32Z","title":"API-Bank: A Comprehensive Benchmark for Tool-Augmented LLMs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.08244","snapshot_observed_at":"2026-08-06T22:02:39.336321Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.22853","last_updated":"2025-07-02T07:55:09Z","snapshot_observed_at":"2026-08-06T21:54:49.869079Z","submitted_at":"2025-06-28T11:28:04Z","title":"DICE-BENCH: Evaluating the Tool-Use Capabilities of Large Language Models in Multi-Round, Multi-Party Dialogues","version":2},"reference_index":21,"source":"arxiv_source","source_observed_at":"2026-08-06T22:02:39.336321Z"},"links":{"cited_paper":"/paper/2304.08244","citing_paper":"/paper/2506.22853"},"observation_digest":"sha256:803c44e14334e36800ee82a64f1cfe8791fd5eb2da13b18bdff8352b330f75ef","observation_id":"a4d59512-941d-4138-8afd-e64b0a39d3d4","resolution":{"observed_at":"2026-08-06T22:02:39.336321Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.08244","last_updated":"2023-10-25T06:54:12Z","snapshot_observed_at":"2026-08-02T00:07:12.855748Z","submitted_at":"2023-04-14T14:05:32Z","title":"API-Bank: A Comprehensive Benchmark for Tool-Augmented LLMs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.08244","snapshot_observed_at":"2026-08-06T21:55:54.130776Z","title":"Api-bank: A comprehensive benchmark for tool-augmented llms,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.23049","last_updated":"2025-06-29T01:13:15Z","snapshot_observed_at":"2026-08-06T21:49:18.821441Z","submitted_at":"2025-06-29T01:13:15Z","title":"AURA: Agent for Understanding, Reasoning, and Automated Tool Use in Voice-Driven Tasks","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-06T21:55:54.130776Z"},"links":{"cited_paper":"/paper/2304.08244","citing_paper":"/paper/2506.23049"},"observation_digest":"sha256:fab72a70382374bcff430c0fd6ba6d434e39978ddf1bc87e6f90b8d9867fc1c7","observation_id":"5e8bd2b4-b095-4eaa-96f2-ef35ab2f289f","resolution":{"observed_at":"2026-08-06T21:55:54.130776Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.08244","last_updated":"2023-10-25T06:54:12Z","snapshot_observed_at":"2026-08-02T00:07:12.855748Z","submitted_at":"2023-04-14T14:05:32Z","title":"API-Bank: A Comprehensive Benchmark for Tool-Augmented LLMs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.08244","snapshot_observed_at":"2026-08-06T21:22:43.768837Z","title":"Api-bank: A comprehensive benchmark for tool-augmented llms // arXiv preprint arXiv:2304.08244","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.00487","last_updated":"2025-07-02T04:35:44Z","snapshot_observed_at":"2026-08-06T21:11:41.074985Z","submitted_at":"2025-07-01T07:02:26Z","title":"MassTool: A Multi-Task Search-Based Tool Retrieval Framework for Large Language Models","version":2},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-06T21:22:43.768837Z"},"links":{"cited_paper":"/paper/2304.08244","citing_paper":"/paper/2507.00487"},"observation_digest":"sha256:88202253ebe0520f708ccbb7c4debc960835b5c1373b9e6a8381238aa0a2f8b4","observation_id":"e53ccb40-cafc-4294-95b5-09604d98476e","resolution":{"observed_at":"2026-08-06T21:22:43.768837Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.08244","last_updated":"2023-10-25T06:54:12Z","snapshot_observed_at":"2026-08-02T00:07:12.855748Z","submitted_at":"2023-04-14T14:05:32Z","title":"API-Bank: A Comprehensive Benchmark for Tool-Augmented LLMs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.08244","snapshot_observed_at":"2026-08-06T17:11:33.897035Z","title":"Api-bank: A comprehensive benchmark for tool-augmented llms","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.11527","last_updated":"2025-07-15T17:56:04Z","snapshot_observed_at":"2026-08-07T07:31:40.303264Z","submitted_at":"2025-07-15T17:56:04Z","title":"DrafterBench: Benchmarking Large Language Models for Tasks Automation in Civil Engineering","version":1},"reference_index":2023,"source":"pdf_text","source_observed_at":"2026-08-06T17:11:33.897035Z"},"links":{"cited_paper":"/paper/2304.08244","citing_paper":"/paper/2507.11527"},"observation_digest":"sha256:51e69b95f99bc50823a07f92054ebe6b69cf36e29f830d03bd7242e5082af36f","observation_id":"c0c88fd4-d6ba-4d6b-90b2-40d720b4d7b9","resolution":{"observed_at":"2026-08-06T17:11:33.897035Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.08244","last_updated":"2023-10-25T06:54:12Z","snapshot_observed_at":"2026-08-02T00:07:12.855748Z","submitted_at":"2023-04-14T14:05:32Z","title":"API-Bank: A Comprehensive Benchmark for Tool-Augmented LLMs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.08244","snapshot_observed_at":"2026-08-06T15:29:47.117662Z","title":"Api-bank: A comprehensive benchmark for tool-augmented llms","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.15761","last_updated":"2025-07-21T16:17:25Z","snapshot_observed_at":"2026-08-06T15:21:47.724683Z","submitted_at":"2025-07-21T16:17:25Z","title":"GasAgent: A Multi-Agent Framework for Automated Gas Optimization in Smart Contracts","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-06T15:29:47.117662Z"},"links":{"cited_paper":"/paper/2304.08244","citing_paper":"/paper/2507.15761"},"observation_digest":"sha256:25134f7488d0d50b1d471d584c2bee9ec5de0b342b13948e0a1c7d2117931837","observation_id":"cc626abe-0d63-4856-9a82-f1567f756248","resolution":{"observed_at":"2026-08-06T15:29:47.117662Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.08244","last_updated":"2023-10-25T06:54:12Z","snapshot_observed_at":"2026-08-02T00:07:12.855748Z","submitted_at":"2023-04-14T14:05:32Z","title":"API-Bank: A Comprehensive Benchmark for Tool-Augmented LLMs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.08244","snapshot_observed_at":"2026-08-06T14:59:37.812022Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.17186","last_updated":"2025-07-31T08:14:21Z","snapshot_observed_at":"2026-08-06T14:52:40.738276Z","submitted_at":"2025-07-23T04:19:16Z","title":"FinGAIA: A Chinese Benchmark for AI Agents in Real-World Financial Domain","version":2},"reference_index":20,"source":"arxiv_source","source_observed_at":"2026-08-06T14:59:37.812022Z"},"links":{"cited_paper":"/paper/2304.08244","citing_paper":"/paper/2507.17186"},"observation_digest":"sha256:9d0a0eafe0f44aeff2c734de472a9b2aa9ce0c55cca4333fd1c4cdf8053834fe","observation_id":"03992d48-a956-461c-9576-8456f8801208","resolution":{"observed_at":"2026-08-06T14:59:37.812022Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.08244","last_updated":"2023-10-25T06:54:12Z","snapshot_observed_at":"2026-08-02T00:07:12.855748Z","submitted_at":"2023-04-14T14:05:32Z","title":"API-Bank: A Comprehensive Benchmark for Tool-Augmented LLMs","version":2},"cited_work":{"arxiv_id":"2304.08244","doi":"10.48550/arxiv.2304.08244","metadata_source":"pith","pith_arxiv_id":"2304.08244","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"API-Bank: A Comprehensive Benchmark for Tool-Augmented LLMs","venue":"cs.CL","work_id":"a20d9332-ab34-485c-a060-1ba47cc98930","year":2023},"citing_paper":{"arxiv_id":"2507.21046","last_updated":"2026-01-16T20:59:08Z","snapshot_observed_at":"2026-08-01T06:32:44.461162Z","submitted_at":"2025-07-28T17:59:05Z","title":"A Survey of Self-Evolving Agents: What, When, How, and Where to Evolve on the Path to Artificial Super Intelligence","version":4},"reference_index":91,"source":"arxiv_source","source_observed_at":"2026-05-14T22:23:14.621091Z"},"links":{"cited_paper":"/paper/2304.08244","citing_paper":"/paper/2507.21046"},"observation_digest":"sha256:eb1463a8bc446fd44cb0aa9605d432ea7983250c2ec5b11bd2b13311d8c49543","observation_id":"29a34b5f-864e-42dd-b212-e88a4fef4cd2","resolution":{"observed_at":"2026-05-15T20:51:41.275571Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-07-14T18:20:23.914543+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-14T18:20:23.914543+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.08244","last_updated":"2023-10-25T06:54:12Z","snapshot_observed_at":"2026-08-02T00:07:12.855748Z","submitted_at":"2023-04-14T14:05:32Z","title":"API-Bank: A Comprehensive Benchmark for Tool-Augmented LLMs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.08244","snapshot_observed_at":"2026-08-06T12:53:45.371850Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.21428","last_updated":"2025-07-29T01:42:06Z","snapshot_observed_at":"2026-08-07T04:48:16.575349Z","submitted_at":"2025-07-29T01:42:06Z","title":"MemTool: Optimizing Short-Term Memory Management for Dynamic Tool Calling in LLM Agent Multi-Turn Conversations","version":1},"reference_index":21,"source":"arxiv_source","source_observed_at":"2026-08-06T12:53:45.371850Z"},"links":{"cited_paper":"/paper/2304.08244","citing_paper":"/paper/2507.21428"},"observation_digest":"sha256:47fbc0d92ef513795d690c613bd6c9747102aa907ad04d4d03925cc72186229d","observation_id":"bf497c9a-95ca-4e9a-b2db-90c07482a652","resolution":{"observed_at":"2026-08-06T12:53:45.371850Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.08244","last_updated":"2023-10-25T06:54:12Z","snapshot_observed_at":"2026-08-02T00:07:12.855748Z","submitted_at":"2023-04-14T14:05:32Z","title":"API-Bank: A Comprehensive Benchmark for Tool-Augmented LLMs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.08244","snapshot_observed_at":"2026-08-06T12:44:21.642196Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.21504","last_updated":"2025-07-29T04:57:02Z","snapshot_observed_at":"2026-08-06T15:35:42.279155Z","submitted_at":"2025-07-29T04:57:02Z","title":"Evaluation and Benchmarking of LLM Agents: A Survey","version":1},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-08-06T12:44:21.642196Z"},"links":{"cited_paper":"/paper/2304.08244","citing_paper":"/paper/2507.21504"},"observation_digest":"sha256:a893daf8fba4f26827e237660a6d614dd10b92664b6ed05f6b6258967ced9d8f","observation_id":"4f5134ab-59be-4e29-8c6c-d1a72f2f2df9","resolution":{"observed_at":"2026-08-06T12:44:21.642196Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.08244","last_updated":"2023-10-25T06:54:12Z","snapshot_observed_at":"2026-08-02T00:07:12.855748Z","submitted_at":"2023-04-14T14:05:32Z","title":"API-Bank: A Comprehensive Benchmark for Tool-Augmented LLMs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.08244","snapshot_observed_at":"2026-08-05T14:49:00.677620Z","title":"arXiv preprint arXiv:2304.08244","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2508.20931","last_updated":"2025-09-01T18:05:06Z","snapshot_observed_at":"2026-08-06T12:07:20.093124Z","submitted_at":"2025-08-28T15:57:33Z","title":"How Can Input Reformulation Improve Tool Usage Accuracy in a Complex Dynamic Environment? A Study on $\\tau$-bench","version":2},"reference_index":2023,"source":"pdf_text","source_observed_at":"2026-08-05T14:49:00.677620Z"},"links":{"cited_paper":"/paper/2304.08244","citing_paper":"/paper/2508.20931"},"observation_digest":"sha256:32dc3b3102d117aa274b1b7e057fa01934a2ce9a4a68a5b45899171ec4228c9a","observation_id":"3701e0ee-3fa4-42a1-a9e1-f2222427131e","resolution":{"observed_at":"2026-08-05T14:49:00.677620Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.08244","last_updated":"2023-10-25T06:54:12Z","snapshot_observed_at":"2026-08-02T00:07:12.855748Z","submitted_at":"2023-04-14T14:05:32Z","title":"API-Bank: A Comprehensive Benchmark for Tool-Augmented LLMs","version":2},"cited_work":{"arxiv_id":"2304.08244","doi":"10.48550/arxiv.2304.08244","metadata_source":"pith","pith_arxiv_id":"2304.08244","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"API-Bank: A Comprehensive Benchmark for Tool-Augmented LLMs","venue":"cs.CL","work_id":"a20d9332-ab34-485c-a060-1ba47cc98930","year":2023},"citing_paper":{"arxiv_id":"2509.18847","last_updated":"2026-04-15T06:57:52Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-09-23T09:35:49Z","title":"Failure Makes the Agent Stronger: Enhancing Accuracy through Structured Reflection for Reliable Tool Interactions","version":3},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-05-18T15:00:51.162221Z"},"links":{"cited_paper":"/paper/2304.08244","citing_paper":"/paper/2509.18847"},"observation_digest":"sha256:12e4fd5d5d419296dfad824a381b6501b8b12349d783f5162c05482be9af0b97","observation_id":"879e3b24-c647-48e2-8516-817f99aa7375","resolution":{"observed_at":"2026-05-18T15:01:31.337055Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-07-14T18:20:23.914543+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-14T18:20:23.914543+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.08244","last_updated":"2023-10-25T06:54:12Z","snapshot_observed_at":"2026-08-02T00:07:12.855748Z","submitted_at":"2023-04-14T14:05:32Z","title":"API-Bank: A Comprehensive Benchmark for Tool-Augmented LLMs","version":2},"cited_work":{"arxiv_id":"2304.08244","doi":"10.48550/arxiv.2304.08244","metadata_source":"pith","pith_arxiv_id":"2304.08244","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"API-Bank: A Comprehensive Benchmark for Tool-Augmented LLMs","venue":"cs.CL","work_id":"a20d9332-ab34-485c-a060-1ba47cc98930","year":2023},"citing_paper":{"arxiv_id":"2510.14703","last_updated":"2026-04-28T18:17:43Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-10-16T14:06:03Z","title":"ToolPRM: Fine-Grained Inference Scaling of Structured Outputs for Function Calling","version":2},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-05-18T06:30:39.858246Z"},"links":{"cited_paper":"/paper/2304.08244","citing_paper":"/paper/2510.14703"},"observation_digest":"sha256:d06fe14b18a7877516cfbaf67c885dd93c12c6d275779df1add3961728617ab5","observation_id":"779cdcf4-e961-4759-9f23-5a1a1158c2f8","resolution":{"observed_at":"2026-05-18T06:30:59.574221Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-07-14T18:20:23.914543+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-14T18:20:23.914543+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.08244","last_updated":"2023-10-25T06:54:12Z","snapshot_observed_at":"2026-08-02T00:07:12.855748Z","submitted_at":"2023-04-14T14:05:32Z","title":"API-Bank: A Comprehensive Benchmark for Tool-Augmented LLMs","version":2},"cited_work":{"arxiv_id":"2304.08244","doi":"10.48550/arxiv.2304.08244","metadata_source":"pith","pith_arxiv_id":"2304.08244","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"API-Bank: A Comprehensive Benchmark for Tool-Augmented LLMs","venue":"cs.CL","work_id":"a20d9332-ab34-485c-a060-1ba47cc98930","year":2023},"citing_paper":{"arxiv_id":"2510.23853","last_updated":"2026-04-15T18:39:35Z","snapshot_observed_at":"2026-07-06T22:34:13.674208Z","submitted_at":"2025-10-27T20:51:58Z","title":"Your LLM Agents are Temporally Blind: The Misalignment Between Tool Use Decisions and Human Time Perception","version":3},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-05-18T03:46:03.228969Z"},"links":{"cited_paper":"/paper/2304.08244","citing_paper":"/paper/2510.23853"},"observation_digest":"sha256:4f8de474841c6344288913de896b43e659958203d15cb1b73982822b375af8b0","observation_id":"a8325615-c5d3-4968-82b9-7a5427425ec7","resolution":{"observed_at":"2026-05-18T03:50:52.576462Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-07-14T18:20:23.914543+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-14T18:20:23.914543+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.08244","last_updated":"2023-10-25T06:54:12Z","snapshot_observed_at":"2026-08-02T00:07:12.855748Z","submitted_at":"2023-04-14T14:05:32Z","title":"API-Bank: A Comprehensive Benchmark for Tool-Augmented LLMs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.08244","snapshot_observed_at":"2026-08-03T21:03:18.066668Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2511.21734","last_updated":"2026-05-23T06:59:38Z","snapshot_observed_at":"2026-08-03T21:03:10.993467Z","submitted_at":"2025-11-21T09:55:34Z","title":"Asking LLMs to Verify First is Almost Free Lunch","version":2},"reference_index":13,"source":"arxiv_source","source_observed_at":"2026-08-03T21:03:18.066668Z"},"links":{"cited_paper":"/paper/2304.08244","citing_paper":"/paper/2511.21734"},"observation_digest":"sha256:e42dbd8bbcf566667544758d3c38a4eaf818ce2d2aa6a3e6a183ac3e91d6a078","observation_id":"a2666c32-226d-4844-8154-8abf2f7506e3","resolution":{"observed_at":"2026-08-03T21:03:18.066668Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.08244","last_updated":"2023-10-25T06:54:12Z","snapshot_observed_at":"2026-08-02T00:07:12.855748Z","submitted_at":"2023-04-14T14:05:32Z","title":"API-Bank: A Comprehensive Benchmark for Tool-Augmented LLMs","version":2},"cited_work":{"arxiv_id":"2304.08244","doi":"10.48550/arxiv.2304.08244","metadata_source":"pith","pith_arxiv_id":"2304.08244","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"API-Bank: A Comprehensive Benchmark for Tool-Augmented LLMs","venue":"cs.CL","work_id":"a20d9332-ab34-485c-a060-1ba47cc98930","year":2023},"citing_paper":{"arxiv_id":"2602.05353","last_updated":"2026-05-03T14:07:13Z","snapshot_observed_at":"2026-08-02T18:14:47.403126Z","submitted_at":"2026-02-05T06:24:15Z","title":"AgentXRay: White-Boxing Agentic Systems via Workflow Reconstruction","version":3},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-05-16T07:38:37.818623Z"},"links":{"cited_paper":"/paper/2304.08244","citing_paper":"/paper/2602.05353"},"observation_digest":"sha256:05e870a9dc6493ea1e5224e8b1e3aa7efaf4fb6a1d46f3b285751dda5da1d703","observation_id":"847dd99e-6431-4773-ac71-84f4d28dc959","resolution":{"observed_at":"2026-05-16T07:40:44.011544Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-07-14T18:20:23.914543+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-14T18:20:23.914543+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.08244","last_updated":"2023-10-25T06:54:12Z","snapshot_observed_at":"2026-08-02T00:07:12.855748Z","submitted_at":"2023-04-14T14:05:32Z","title":"API-Bank: A Comprehensive Benchmark for Tool-Augmented LLMs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.08244","snapshot_observed_at":"2026-08-04T06:09:51.803474Z","title":"API-Bank: A Comprehensive Benchmark for Tool-Augmented LLMs.arXiv preprint arXiv:2304.08244,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2602.07086","last_updated":"2026-08-01T15:58:15Z","snapshot_observed_at":"2026-08-06T23:10:57.704344Z","submitted_at":"2026-02-06T08:37:06Z","title":"RAG Strategies for Natural Language-Based SQL Query and REST API Call Generation","version":2},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-04T06:09:51.803474Z"},"links":{"cited_paper":"/paper/2304.08244","citing_paper":"/paper/2602.07086"},"observation_digest":"sha256:1d5906d4ca960d10e5e5173541fd76912e76ea5f6ebbd41917f1ac09a2afebab","observation_id":"a4d50127-be3c-47ff-98a4-5d172e8a2203","resolution":{"observed_at":"2026-08-04T06:09:51.803474Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.08244","last_updated":"2023-10-25T06:54:12Z","snapshot_observed_at":"2026-08-02T00:07:12.855748Z","submitted_at":"2023-04-14T14:05:32Z","title":"API-Bank: A Comprehensive Benchmark for Tool-Augmented LLMs","version":2},"cited_work":{"arxiv_id":"2304.08244","doi":"10.48550/arxiv.2304.08244","metadata_source":"pith","pith_arxiv_id":"2304.08244","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"API-Bank: A Comprehensive Benchmark for Tool-Augmented LLMs","venue":"cs.CL","work_id":"a20d9332-ab34-485c-a060-1ba47cc98930","year":2023},"citing_paper":{"arxiv_id":"2602.11224","last_updated":"2026-04-28T16:08:25Z","snapshot_observed_at":"2026-07-06T22:45:29.609382Z","submitted_at":"2026-02-11T13:31:18Z","title":"Agent-Diff: Benchmarking LLM Agents on Enterprise API Tasks via Code Execution with State-Diff-Based Evaluation","version":3},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-05-16T03:04:17.755968Z"},"links":{"cited_paper":"/paper/2304.08244","citing_paper":"/paper/2602.11224"},"observation_digest":"sha256:97d1966c50d9163db619f65175f95e886d2de19a1c3c1fc9ae04d9a366aeaa9b","observation_id":"ab1047b2-a35d-471f-8661-f38d8ed2d744","resolution":{"observed_at":"2026-05-16T03:07:11.528454Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-07-14T18:20:23.914543+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-14T18:20:23.914543+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.08244","last_updated":"2023-10-25T06:54:12Z","snapshot_observed_at":"2026-08-02T00:07:12.855748Z","submitted_at":"2023-04-14T14:05:32Z","title":"API-Bank: A Comprehensive Benchmark for Tool-Augmented LLMs","version":2},"cited_work":{"arxiv_id":"2304.08244","doi":"10.48550/arxiv.2304.08244","metadata_source":"pith","pith_arxiv_id":"2304.08244","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"API-Bank: A Comprehensive Benchmark for Tool-Augmented LLMs","venue":"cs.CL","work_id":"a20d9332-ab34-485c-a060-1ba47cc98930","year":2023},"citing_paper":{"arxiv_id":"2602.21265","last_updated":"2026-05-18T08:26:18Z","snapshot_observed_at":"2026-08-02T06:00:00.618001Z","submitted_at":"2026-02-24T09:23:12Z","title":"ToolMATH: A Diagnostic Benchmark for Long-Horizon Tool Use under Systematic Tool-Catalog Constraints","version":2},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-07-11T11:50:26.030339Z"},"links":{"cited_paper":"/paper/2304.08244","citing_paper":"/paper/2602.21265"},"observation_digest":"sha256:32c27295e5d7c0964f23e903885c2e9d9f569961659d40002630b255fc24992e","observation_id":"636fa33c-ccd1-465e-82ac-8d3600073afe","resolution":{"observed_at":"2026-05-21T12:00:04.121440Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-07-14T18:20:23.914543+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-14T18:20:23.914543+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.08244","last_updated":"2023-10-25T06:54:12Z","snapshot_observed_at":"2026-08-02T00:07:12.855748Z","submitted_at":"2023-04-14T14:05:32Z","title":"API-Bank: A Comprehensive Benchmark for Tool-Augmented LLMs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.08244","snapshot_observed_at":"2026-08-04T05:54:53.390032Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2603.08262","last_updated":"2026-08-01T16:26:47Z","snapshot_observed_at":"2026-08-06T23:45:46.344090Z","submitted_at":"2026-03-09T11:33:05Z","title":"FinToolBench: Evaluating LLM Agents for Real-World Financial Tool Use","version":2},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-04T05:54:53.390032Z"},"links":{"cited_paper":"/paper/2304.08244","citing_paper":"/paper/2603.08262"},"observation_digest":"sha256:6cd82bf44e6f27bfa7e7da311d05e5e57f8ea2d9649778ae8ac87c14fd462743","observation_id":"f50ff977-8629-4a1f-a475-43247594d8a6","resolution":{"observed_at":"2026-08-04T05:54:53.390032Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.08244","last_updated":"2023-10-25T06:54:12Z","snapshot_observed_at":"2026-08-02T00:07:12.855748Z","submitted_at":"2023-04-14T14:05:32Z","title":"API-Bank: A Comprehensive Benchmark for Tool-Augmented LLMs","version":2},"cited_work":{"arxiv_id":"2304.08244","doi":"10.48550/arxiv.2304.08244","metadata_source":"pith","pith_arxiv_id":"2304.08244","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"API-Bank: A Comprehensive Benchmark for Tool-Augmented LLMs","venue":"cs.CL","work_id":"a20d9332-ab34-485c-a060-1ba47cc98930","year":2023},"citing_paper":{"arxiv_id":"2603.14987","last_updated":"2026-05-21T06:24:06Z","snapshot_observed_at":"2026-08-03T21:30:30.993382Z","submitted_at":"2026-03-16T08:51:33Z","title":"Beyond Benchmark Islands: Toward Representative Trustworthiness Evaluation for Agentic AI","version":2},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-05-22T10:19:56.003219Z"},"links":{"cited_paper":"/paper/2304.08244","citing_paper":"/paper/2603.14987"},"observation_digest":"sha256:5c03048102e9e25df33241875f0db50503dbe23e4e1ac13a8baf66f0f2e48245","observation_id":"d10a8c0c-04cd-4db8-9426-48fa64c28d0f","resolution":{"observed_at":"2026-05-22T10:21:23.204138Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-07-14T18:20:23.914543+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-14T18:20:23.914543+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.08244","last_updated":"2023-10-25T06:54:12Z","snapshot_observed_at":"2026-08-02T00:07:12.855748Z","submitted_at":"2023-04-14T14:05:32Z","title":"API-Bank: A Comprehensive Benchmark for Tool-Augmented LLMs","version":2},"cited_work":{"arxiv_id":"2304.08244","doi":"10.48550/arxiv.2304.08244","metadata_source":"pith","pith_arxiv_id":"2304.08244","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"API-Bank: A Comprehensive Benchmark for Tool-Augmented LLMs","venue":"cs.CL","work_id":"a20d9332-ab34-485c-a060-1ba47cc98930","year":2023},"citing_paper":{"arxiv_id":"2604.07551","last_updated":"2026-04-08T19:53:26Z","snapshot_observed_at":"2026-07-06T22:55:46.307881Z","submitted_at":"2026-04-08T19:53:26Z","title":"MCP-DPT: A Defense-Placement Taxonomy and Coverage Analysis for Model Context Protocol Security","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-05-10T17:10:50.283791Z"},"links":{"cited_paper":"/paper/2304.08244","citing_paper":"/paper/2604.07551"},"observation_digest":"sha256:ac664549610d8d2eb2715566941227598d857e035a8c761f095300eaadf4ac97","observation_id":"ca9df927-aee1-41ec-bee8-957e6d3e6a7a","resolution":{"observed_at":"2026-05-15T20:51:41.275571Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-07-14T18:20:23.914543+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-14T18:20:23.914543+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.08244","last_updated":"2023-10-25T06:54:12Z","snapshot_observed_at":"2026-08-02T00:07:12.855748Z","submitted_at":"2023-04-14T14:05:32Z","title":"API-Bank: A Comprehensive Benchmark for Tool-Augmented LLMs","version":2},"cited_work":{"arxiv_id":"2304.08244","doi":"10.48550/arxiv.2304.08244","metadata_source":"pith","pith_arxiv_id":"2304.08244","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"API-Bank: A Comprehensive Benchmark for Tool-Augmented LLMs","venue":"cs.CL","work_id":"a20d9332-ab34-485c-a060-1ba47cc98930","year":2023},"citing_paper":{"arxiv_id":"2604.09285","last_updated":"2026-04-10T12:55:23Z","snapshot_observed_at":"2026-08-05T01:20:50.277571Z","submitted_at":"2026-04-10T12:55:23Z","title":"SAGE: A Service Agent Graph-guided Evaluation Benchmark","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-05-10T16:41:23.956104Z"},"links":{"cited_paper":"/paper/2304.08244","citing_paper":"/paper/2604.09285"},"observation_digest":"sha256:40c334f090f5f3c20b26771c4e969cbc10657186bf002df6ede6f5ef750c8f2f","observation_id":"0061d9e7-b0f0-4fd1-81d2-ae2e51da7d0c","resolution":{"observed_at":"2026-05-15T20:51:41.275571Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-07-14T18:20:23.914543+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-14T18:20:23.914543+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.08244","last_updated":"2023-10-25T06:54:12Z","snapshot_observed_at":"2026-08-02T00:07:12.855748Z","submitted_at":"2023-04-14T14:05:32Z","title":"API-Bank: A Comprehensive Benchmark for Tool-Augmented LLMs","version":2},"cited_work":{"arxiv_id":"2304.08244","doi":"10.48550/arxiv.2304.08244","metadata_source":"pith","pith_arxiv_id":"2304.08244","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"API-Bank: A Comprehensive Benchmark for Tool-Augmented LLMs","venue":"cs.CL","work_id":"a20d9332-ab34-485c-a060-1ba47cc98930","year":2023},"citing_paper":{"arxiv_id":"2604.12259","last_updated":"2026-04-14T04:23:32Z","snapshot_observed_at":"2026-07-30T11:21:02.274476Z","submitted_at":"2026-04-14T04:23:32Z","title":"A Periodic Space of Distributed Computing: Vision & Framework","version":1},"reference_index":77,"source":"pdf_text","source_observed_at":"2026-05-10T15:55:44.933926Z"},"links":{"cited_paper":"/paper/2304.08244","citing_paper":"/paper/2604.12259"},"observation_digest":"sha256:92a69c6c073ccc9c3f9ba9d279f7a14c7554353c2ca43eb0f4911953a05b493b","observation_id":"2a58aa53-804e-484f-9e7b-1ff3a98a0502","resolution":{"observed_at":"2026-05-15T20:51:41.275571Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-07-14T18:20:23.914543+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-14T18:20:23.914543+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.08244","last_updated":"2023-10-25T06:54:12Z","snapshot_observed_at":"2026-08-02T00:07:12.855748Z","submitted_at":"2023-04-14T14:05:32Z","title":"API-Bank: A Comprehensive Benchmark for Tool-Augmented LLMs","version":2},"cited_work":{"arxiv_id":"2304.08244","doi":"10.48550/arxiv.2304.08244","metadata_source":"pith","pith_arxiv_id":"2304.08244","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"API-Bank: A Comprehensive Benchmark for Tool-Augmented LLMs","venue":"cs.CL","work_id":"a20d9332-ab34-485c-a060-1ba47cc98930","year":2023},"citing_paper":{"arxiv_id":"2604.13286","last_updated":"2026-04-14T20:26:34Z","snapshot_observed_at":"2026-08-02T11:31:41.800497Z","submitted_at":"2026-04-14T20:26:34Z","title":"English is Not All You Need: Systematically Exploring the Role of Multilinguality in LLM Post-Training","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-05-10T15:08:32.517020Z"},"links":{"cited_paper":"/paper/2304.08244","citing_paper":"/paper/2604.13286"},"observation_digest":"sha256:389d9844bccac7df4c0d5a0ac8b188c86fd115ead57169a3f924d7b4ba1b8808","observation_id":"9e87fac9-b753-4da4-8aab-a8a075835d1d","resolution":{"observed_at":"2026-05-15T20:51:41.275571Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-07-14T18:20:23.914543+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-14T18:20:23.914543+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.08244","last_updated":"2023-10-25T06:54:12Z","snapshot_observed_at":"2026-08-02T00:07:12.855748Z","submitted_at":"2023-04-14T14:05:32Z","title":"API-Bank: A Comprehensive Benchmark for Tool-Augmented LLMs","version":2},"cited_work":{"arxiv_id":"2304.08244","doi":"10.48550/arxiv.2304.08244","metadata_source":"pith","pith_arxiv_id":"2304.08244","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"API-Bank: A Comprehensive Benchmark for Tool-Augmented LLMs","venue":"cs.CL","work_id":"a20d9332-ab34-485c-a060-1ba47cc98930","year":2023},"citing_paper":{"arxiv_id":"2604.17870","last_updated":"2026-04-20T06:31:11Z","snapshot_observed_at":"2026-07-06T23:04:50.935335Z","submitted_at":"2026-04-20T06:31:11Z","title":"GraSP: Graph-Structured Skill Compositions for LLM Agents","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-10T04:16:48.625528Z"},"links":{"cited_paper":"/paper/2304.08244","citing_paper":"/paper/2604.17870"},"observation_digest":"sha256:7a3669caababa8a3da10364865280c643ba659f0b7b7b77fa6879a35d681c506","observation_id":"9cfb58f7-7bbf-4682-bdba-238f0af61c55","resolution":{"observed_at":"2026-05-15T20:51:41.275571Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-07-14T18:20:23.914543+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-14T18:20:23.914543+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.08244","last_updated":"2023-10-25T06:54:12Z","snapshot_observed_at":"2026-08-02T00:07:12.855748Z","submitted_at":"2023-04-14T14:05:32Z","title":"API-Bank: A Comprehensive Benchmark for Tool-Augmented LLMs","version":2},"cited_work":{"arxiv_id":"2304.08244","doi":"10.48550/arxiv.2304.08244","metadata_source":"pith","pith_arxiv_id":"2304.08244","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"API-Bank: A Comprehensive Benchmark for Tool-Augmented LLMs","venue":"cs.CL","work_id":"a20d9332-ab34-485c-a060-1ba47cc98930","year":2023},"citing_paper":{"arxiv_id":"2604.18327","last_updated":"2026-04-20T14:29:08Z","snapshot_observed_at":"2026-07-06T23:05:13.178333Z","submitted_at":"2026-04-20T14:29:08Z","title":"PARM: Pipeline-Adapted Reward Model","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-05-10T05:15:26.015817Z"},"links":{"cited_paper":"/paper/2304.08244","citing_paper":"/paper/2604.18327"},"observation_digest":"sha256:8a125c18903c12e9052b1e54b69d3ea36162c055e3147a218c03b1f953c310f5","observation_id":"5c724b56-4960-4e53-b244-a3d4218aadd9","resolution":{"observed_at":"2026-05-15T20:51:41.275571Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-07-14T18:20:23.914543+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-14T18:20:23.914543+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.08244","last_updated":"2023-10-25T06:54:12Z","snapshot_observed_at":"2026-08-02T00:07:12.855748Z","submitted_at":"2023-04-14T14:05:32Z","title":"API-Bank: A Comprehensive Benchmark for Tool-Augmented LLMs","version":2},"cited_work":{"arxiv_id":"2304.08244","doi":"10.48550/arxiv.2304.08244","metadata_source":"pith","pith_arxiv_id":"2304.08244","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"API-Bank: A Comprehensive Benchmark for Tool-Augmented LLMs","venue":"cs.CL","work_id":"a20d9332-ab34-485c-a060-1ba47cc98930","year":2023},"citing_paper":{"arxiv_id":"2604.19667","last_updated":"2026-05-26T16:14:10Z","snapshot_observed_at":"2026-08-01T19:03:44.336277Z","submitted_at":"2026-04-21T16:49:11Z","title":"Chat2Workflow: A Benchmark for Generating Executable Visual Workflows with Natural Language","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-10T02:56:53.055513Z"},"links":{"cited_paper":"/paper/2304.08244","citing_paper":"/paper/2604.19667"},"observation_digest":"sha256:78ebdf349c3f1b5afedb5758f527bc718a5bcebc9047d6b26e4378c842d7d3ea","observation_id":"3ebb8397-8721-4bde-a128-e3f0b428e1dc","resolution":{"observed_at":"2026-05-15T20:51:41.275571Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-07-14T18:20:23.914543+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-14T18:20:23.914543+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.08244","last_updated":"2023-10-25T06:54:12Z","snapshot_observed_at":"2026-08-02T00:07:12.855748Z","submitted_at":"2023-04-14T14:05:32Z","title":"API-Bank: A Comprehensive Benchmark for Tool-Augmented LLMs","version":2},"cited_work":{"arxiv_id":"2304.08244","doi":"10.48550/arxiv.2304.08244","metadata_source":"pith","pith_arxiv_id":"2304.08244","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"API-Bank: A Comprehensive Benchmark for Tool-Augmented LLMs","venue":"cs.CL","work_id":"a20d9332-ab34-485c-a060-1ba47cc98930","year":2023},"citing_paper":{"arxiv_id":"2604.20148","last_updated":"2026-04-22T03:25:17Z","snapshot_observed_at":"2026-07-06T23:06:40.154278Z","submitted_at":"2026-04-22T03:25:17Z","title":"Meta-Tool: Efficient Few-Shot Tool Adaptation for Small Language Models","version":1},"reference_index":39,"source":"arxiv_source","source_observed_at":"2026-05-10T00:50:27.257888Z"},"links":{"cited_paper":"/paper/2304.08244","citing_paper":"/paper/2604.20148"},"observation_digest":"sha256:3f34a8d9118a6d6ee9010e95de601634f9dc825fd8c4815b2d45647eb27a682b","observation_id":"93c2ddaa-9498-4fb2-b04f-ec25359d1462","resolution":{"observed_at":"2026-05-15T20:51:41.275571Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-07-14T18:20:23.914543+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-14T18:20:23.914543+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.08244","last_updated":"2023-10-25T06:54:12Z","snapshot_observed_at":"2026-08-02T00:07:12.855748Z","submitted_at":"2023-04-14T14:05:32Z","title":"API-Bank: A Comprehensive Benchmark for Tool-Augmented LLMs","version":2},"cited_work":{"arxiv_id":"2304.08244","doi":"10.48550/arxiv.2304.08244","metadata_source":"pith","pith_arxiv_id":"2304.08244","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"API-Bank: A Comprehensive Benchmark for Tool-Augmented LLMs","venue":"cs.CL","work_id":"a20d9332-ab34-485c-a060-1ba47cc98930","year":2023},"citing_paper":{"arxiv_id":"2604.22760","last_updated":"2026-03-09T20:28:24Z","snapshot_observed_at":"2026-07-31T19:38:35.982654Z","submitted_at":"2026-03-09T20:28:24Z","title":"Quantifying Divergence in Inter-LLM Communication Through API Retrieval and Ranking","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-15T12:59:56.133502Z"},"links":{"cited_paper":"/paper/2304.08244","citing_paper":"/paper/2604.22760"},"observation_digest":"sha256:950686c8526e0f3568cb1ee6d6f9eb65ada838d30c5352ac5baa96cedbb08995","observation_id":"239173b7-1c36-4516-91c7-d301940f608e","resolution":{"observed_at":"2026-05-15T20:51:41.275571Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-07-14T18:20:23.914543+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-14T18:20:23.914543+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.08244","last_updated":"2023-10-25T06:54:12Z","snapshot_observed_at":"2026-08-02T00:07:12.855748Z","submitted_at":"2023-04-14T14:05:32Z","title":"API-Bank: A Comprehensive Benchmark for Tool-Augmented LLMs","version":2},"cited_work":{"arxiv_id":"2304.08244","doi":"10.48550/arxiv.2304.08244","metadata_source":"pith","pith_arxiv_id":"2304.08244","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"API-Bank: A Comprehensive Benchmark for Tool-Augmented LLMs","venue":"cs.CL","work_id":"a20d9332-ab34-485c-a060-1ba47cc98930","year":2023},"citing_paper":{"arxiv_id":"2604.28139","last_updated":"2026-05-01T09:39:37Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2026-04-30T17:23:19Z","title":"Claw-Eval-Live: A Live Agent Benchmark for Evolving Real-World Workflows","version":2},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-05-07T05:52:13.100867Z"},"links":{"cited_paper":"/paper/2304.08244","citing_paper":"/paper/2604.28139"},"observation_digest":"sha256:f8f010cab37021f0dc4e1ece23dc063cf903dcb3086a5f74a071f134d05193ae","observation_id":"759fa504-1109-4f66-8091-e3684506dbd9","resolution":{"observed_at":"2026-05-15T20:51:41.275571Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-07-14T18:20:23.914543+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-14T18:20:23.914543+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.08244","last_updated":"2023-10-25T06:54:12Z","snapshot_observed_at":"2026-08-02T00:07:12.855748Z","submitted_at":"2023-04-14T14:05:32Z","title":"API-Bank: A Comprehensive Benchmark for Tool-Augmented LLMs","version":2},"cited_work":{"arxiv_id":"2304.08244","doi":"10.48550/arxiv.2304.08244","metadata_source":"pith","pith_arxiv_id":"2304.08244","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"API-Bank: A Comprehensive Benchmark for Tool-Augmented LLMs","venue":"cs.CL","work_id":"a20d9332-ab34-485c-a060-1ba47cc98930","year":2023},"citing_paper":{"arxiv_id":"2605.07990","last_updated":"2026-05-08T16:47:08Z","snapshot_observed_at":"2026-08-02T18:36:12.949377Z","submitted_at":"2026-05-08T16:47:08Z","title":"Tool Calling is Linearly Readable and Steerable in Language Models","version":1},"reference_index":56,"source":"arxiv_source","source_observed_at":"2026-05-11T03:09:11.013914Z"},"links":{"cited_paper":"/paper/2304.08244","citing_paper":"/paper/2605.07990"},"observation_digest":"sha256:1a969d11d2e377caacb0ce576bd99ac17db6ebc61c1135809f75c32341429514","observation_id":"fd25fa72-8feb-46a5-b24e-4d9c62576748","resolution":{"observed_at":"2026-05-15T20:51:41.275571Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-07-14T18:20:23.914543+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-14T18:20:23.914543+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.08244","last_updated":"2023-10-25T06:54:12Z","snapshot_observed_at":"2026-08-02T00:07:12.855748Z","submitted_at":"2023-04-14T14:05:32Z","title":"API-Bank: A Comprehensive Benchmark for Tool-Augmented LLMs","version":2},"cited_work":{"arxiv_id":"2304.08244","doi":"10.48550/arxiv.2304.08244","metadata_source":"pith","pith_arxiv_id":"2304.08244","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"API-Bank: A Comprehensive Benchmark for Tool-Augmented LLMs","venue":"cs.CL","work_id":"a20d9332-ab34-485c-a060-1ba47cc98930","year":2023},"citing_paper":{"arxiv_id":"2605.09544","last_updated":"2026-05-10T13:56:46Z","snapshot_observed_at":"2026-07-31T07:40:23.585067Z","submitted_at":"2026-05-10T13:56:46Z","title":"TIDE-Bench: Task-Aware and Diagnostic Evaluation of Tool-Integrated Reasoning","version":1},"reference_index":17,"source":"arxiv_source","source_observed_at":"2026-05-12T02:47:31.830410Z"},"links":{"cited_paper":"/paper/2304.08244","citing_paper":"/paper/2605.09544"},"observation_digest":"sha256:0538c15bf88b8ba134db7ff47bfb340b81312e69c70c434ee48b54c0de4faf5e","observation_id":"54975448-a0e9-48ef-93e9-4394a9ad97bb","resolution":{"observed_at":"2026-05-15T20:51:41.275571Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-07-14T18:20:23.914543+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-14T18:20:23.914543+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.08244","last_updated":"2023-10-25T06:54:12Z","snapshot_observed_at":"2026-08-02T00:07:12.855748Z","submitted_at":"2023-04-14T14:05:32Z","title":"API-Bank: A Comprehensive Benchmark for Tool-Augmented LLMs","version":2},"cited_work":{"arxiv_id":"2304.08244","doi":"10.48550/arxiv.2304.08244","metadata_source":"pith","pith_arxiv_id":"2304.08244","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"API-Bank: A Comprehensive Benchmark for Tool-Augmented LLMs","venue":"cs.CL","work_id":"a20d9332-ab34-485c-a060-1ba47cc98930","year":2023},"citing_paper":{"arxiv_id":"2605.09734","last_updated":"2026-05-10T20:09:41Z","snapshot_observed_at":"2026-07-06T23:21:49.499951Z","submitted_at":"2026-05-10T20:09:41Z","title":"Trajectory Supervision for Continual Tool-Use Learning in LLMs","version":1},"reference_index":4,"source":"arxiv_source","source_observed_at":"2026-05-12T03:20:20.779419Z"},"links":{"cited_paper":"/paper/2304.08244","citing_paper":"/paper/2605.09734"},"observation_digest":"sha256:4228f36fb2be0363bbe63677e0c98e935f1969ee47bd491e48f7f95b97bdc7f1","observation_id":"0b97cd65-7bea-4e58-a6b0-e3668a8e6abf","resolution":{"observed_at":"2026-05-15T20:51:41.275571Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-07-14T18:20:23.914543+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-14T18:20:23.914543+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.08244","last_updated":"2023-10-25T06:54:12Z","snapshot_observed_at":"2026-08-02T00:07:12.855748Z","submitted_at":"2023-04-14T14:05:32Z","title":"API-Bank: A Comprehensive Benchmark for Tool-Augmented LLMs","version":2},"cited_work":{"arxiv_id":"2304.08244","doi":"10.48550/arxiv.2304.08244","metadata_source":"pith","pith_arxiv_id":"2304.08244","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"API-Bank: A Comprehensive Benchmark for Tool-Augmented LLMs","venue":"cs.CL","work_id":"a20d9332-ab34-485c-a060-1ba47cc98930","year":2023},"citing_paper":{"arxiv_id":"2605.14038","last_updated":"2026-05-17T15:23:37Z","snapshot_observed_at":"2026-08-03T15:23:23.551096Z","submitted_at":"2026-05-13T18:59:28Z","title":"Model-Adaptive Tool Necessity Reveals the Knowing-Doing Gap in LLM Tool Use","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-05-15T05:31:54.252816Z"},"links":{"cited_paper":"/paper/2304.08244","citing_paper":"/paper/2605.14038"},"observation_digest":"sha256:19ac64c734b14c1fab09242f027e1315badc6c4316f556dbc54b6b4c61316235","observation_id":"2209d1b2-d772-49b3-8ae0-03fa309fffde","resolution":{"observed_at":"2026-05-15T20:51:41.275571Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-07-14T18:20:23.914543+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-14T18:20:23.914543+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.08244","last_updated":"2023-10-25T06:54:12Z","snapshot_observed_at":"2026-08-02T00:07:12.855748Z","submitted_at":"2023-04-14T14:05:32Z","title":"API-Bank: A Comprehensive Benchmark for Tool-Augmented LLMs","version":2},"cited_work":{"arxiv_id":"2304.08244","doi":"10.48550/arxiv.2304.08244","metadata_source":"pith","pith_arxiv_id":"2304.08244","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"API-Bank: A Comprehensive Benchmark for Tool-Augmented LLMs","venue":"cs.CL","work_id":"a20d9332-ab34-485c-a060-1ba47cc98930","year":2023},"citing_paper":{"arxiv_id":"2605.14038","last_updated":"2026-05-17T15:23:37Z","snapshot_observed_at":"2026-08-03T15:23:23.551096Z","submitted_at":"2026-05-13T18:59:28Z","title":"Model-Adaptive Tool Necessity Reveals the Knowing-Doing Gap in LLM Tool Use","version":2},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-05-20T20:52:20.975459Z"},"links":{"cited_paper":"/paper/2304.08244","citing_paper":"/paper/2605.14038"},"observation_digest":"sha256:fc6ea5726e6eb2cd9eea38949438d632daba37ca8dd8823a50430823c17daecf","observation_id":"5b7dde3c-31f3-4ae2-8ac3-39aae584ba88","resolution":{"observed_at":"2026-05-20T20:53:43.544451Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-07-14T18:20:23.914543+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-14T18:20:23.914543+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.08244","last_updated":"2023-10-25T06:54:12Z","snapshot_observed_at":"2026-08-02T00:07:12.855748Z","submitted_at":"2023-04-14T14:05:32Z","title":"API-Bank: A Comprehensive Benchmark for Tool-Augmented LLMs","version":2},"cited_work":{"arxiv_id":"2304.08244","doi":"10.48550/arxiv.2304.08244","metadata_source":"pith","pith_arxiv_id":"2304.08244","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"API-Bank: A Comprehensive Benchmark for Tool-Augmented LLMs","venue":"cs.CL","work_id":"a20d9332-ab34-485c-a060-1ba47cc98930","year":2023},"citing_paper":{"arxiv_id":"2605.16508","last_updated":"2026-05-15T18:05:21Z","snapshot_observed_at":"2026-08-03T01:50:34.102829Z","submitted_at":"2026-05-15T18:05:21Z","title":"The Scaling Laws of Skills in LLM Agent Systems","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-05-20T18:10:08.737710Z"},"links":{"cited_paper":"/paper/2304.08244","citing_paper":"/paper/2605.16508"},"observation_digest":"sha256:bed29a98c27077a1acdc8389ac8220c7ca1722113df1b1f3923512603f8b3879","observation_id":"6c47fd8d-faee-444e-8d3f-da9f3b66d61f","resolution":{"observed_at":"2026-05-20T18:13:37.648430Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-07-14T18:20:23.914543+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-14T18:20:23.914543+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.08244","last_updated":"2023-10-25T06:54:12Z","snapshot_observed_at":"2026-08-02T00:07:12.855748Z","submitted_at":"2023-04-14T14:05:32Z","title":"API-Bank: A Comprehensive Benchmark for Tool-Augmented LLMs","version":2},"cited_work":{"arxiv_id":"2304.08244","doi":"10.48550/arxiv.2304.08244","metadata_source":"pith","pith_arxiv_id":"2304.08244","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"API-Bank: A Comprehensive Benchmark for Tool-Augmented LLMs","venue":"cs.CL","work_id":"a20d9332-ab34-485c-a060-1ba47cc98930","year":2023},"citing_paper":{"arxiv_id":"2605.17558","last_updated":"2026-05-17T17:38:17Z","snapshot_observed_at":"2026-08-03T01:04:03.264088Z","submitted_at":"2026-05-17T17:38:17Z","title":"Firefly: Illuminating Large-Scale Verified Tool-Call Data Generation from Real APIs","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-05-19T22:30:21.756864Z"},"links":{"cited_paper":"/paper/2304.08244","citing_paper":"/paper/2605.17558"},"observation_digest":"sha256:6bd27324957e724037f2b5eb0f5d5bd4816d982432a258faf900fc4a7de09134","observation_id":"90dc69b7-7f6e-4b53-aff8-4fe5b1b3a533","resolution":{"observed_at":"2026-05-19T22:32:49.692323Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-07-14T18:20:23.914543+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-14T18:20:23.914543+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.08244","last_updated":"2023-10-25T06:54:12Z","snapshot_observed_at":"2026-08-02T00:07:12.855748Z","submitted_at":"2023-04-14T14:05:32Z","title":"API-Bank: A Comprehensive Benchmark for Tool-Augmented LLMs","version":2},"cited_work":{"arxiv_id":"2304.08244","doi":"10.48550/arxiv.2304.08244","metadata_source":"pith","pith_arxiv_id":"2304.08244","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"API-Bank: A Comprehensive Benchmark for Tool-Augmented LLMs","venue":"cs.CL","work_id":"a20d9332-ab34-485c-a060-1ba47cc98930","year":2023},"citing_paper":{"arxiv_id":"2605.17774","last_updated":"2026-05-26T01:46:08Z","snapshot_observed_at":"2026-07-06T23:28:45.646975Z","submitted_at":"2026-05-18T02:48:46Z","title":"Internalizing Tool Knowledge in Small Language Models via QLoRA Fine-Tuning","version":2},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-06-30T19:08:57.644677Z"},"links":{"cited_paper":"/paper/2304.08244","citing_paper":"/paper/2605.17774"},"observation_digest":"sha256:aa4ff6113150199f68932116cff7968fe71e2b7e990f51088a691207b7842e59","observation_id":"c3c71dd9-7383-498e-8b7b-17c10c0a095a","resolution":{"observed_at":"2026-06-30T19:15:00.685356Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-07-14T18:20:23.914543+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-14T18:20:23.914543+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.08244","last_updated":"2023-10-25T06:54:12Z","snapshot_observed_at":"2026-08-02T00:07:12.855748Z","submitted_at":"2023-04-14T14:05:32Z","title":"API-Bank: A Comprehensive Benchmark for Tool-Augmented LLMs","version":2},"cited_work":{"arxiv_id":"2304.08244","doi":"10.48550/arxiv.2304.08244","metadata_source":"pith","pith_arxiv_id":"2304.08244","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"API-Bank: A Comprehensive Benchmark for Tool-Augmented LLMs","venue":"cs.CL","work_id":"a20d9332-ab34-485c-a060-1ba47cc98930","year":2023},"citing_paper":{"arxiv_id":"2605.18133","last_updated":"2026-05-18T09:38:18Z","snapshot_observed_at":"2026-07-06T23:29:04.241654Z","submitted_at":"2026-05-18T09:38:18Z","title":"An Empirical Study of Privacy Leakage Chains via Prompt Injection in Black-Box Chatbot Environments","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-05-20T09:58:05.349147Z"},"links":{"cited_paper":"/paper/2304.08244","citing_paper":"/paper/2605.18133"},"observation_digest":"sha256:26bd066d0f291e0d0601c1a8fa9f73d00ff892547d6aa9a0461633e41e2463ba","observation_id":"d230f240-1a2f-4e96-811f-2bde4f1dacd8","resolution":{"observed_at":"2026-05-20T09:58:10.845623Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-07-14T18:20:23.914543+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-14T18:20:23.914543+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.08244","last_updated":"2023-10-25T06:54:12Z","snapshot_observed_at":"2026-08-02T00:07:12.855748Z","submitted_at":"2023-04-14T14:05:32Z","title":"API-Bank: A Comprehensive Benchmark for Tool-Augmented LLMs","version":2},"cited_work":{"arxiv_id":"2304.08244","doi":"10.48550/arxiv.2304.08244","metadata_source":"pith","pith_arxiv_id":"2304.08244","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"API-Bank: A Comprehensive Benchmark for Tool-Augmented LLMs","venue":"cs.CL","work_id":"a20d9332-ab34-485c-a060-1ba47cc98930","year":2023},"citing_paper":{"arxiv_id":"2605.23574","last_updated":"2026-05-22T12:44:01Z","snapshot_observed_at":"2026-07-06T23:33:44.335185Z","submitted_at":"2026-05-22T12:44:01Z","title":"Push Your Agent: Measuring and Enforcing Quantitative Goal Persistence in Long-Horizon LLM Agents","version":1},"reference_index":18,"source":"arxiv_source","source_observed_at":"2026-05-25T04:47:01.153482Z"},"links":{"cited_paper":"/paper/2304.08244","citing_paper":"/paper/2605.23574"},"observation_digest":"sha256:36d85e465cadc0a3560b3f83580f3b047879a245a0710d4c0f98003f41615aa9","observation_id":"f9a0d735-9023-45df-ba13-1427bfd3b862","resolution":{"observed_at":"2026-05-25T04:50:20.462079Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-07-14T18:20:23.914543+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-14T18:20:23.914543+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.08244","last_updated":"2023-10-25T06:54:12Z","snapshot_observed_at":"2026-08-02T00:07:12.855748Z","submitted_at":"2023-04-14T14:05:32Z","title":"API-Bank: A Comprehensive Benchmark for Tool-Augmented LLMs","version":2},"cited_work":{"arxiv_id":"2304.08244","doi":"10.48550/arxiv.2304.08244","metadata_source":"pith","pith_arxiv_id":"2304.08244","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"API-Bank: A Comprehensive Benchmark for Tool-Augmented LLMs","venue":"cs.CL","work_id":"a20d9332-ab34-485c-a060-1ba47cc98930","year":2023},"citing_paper":{"arxiv_id":"2605.26521","last_updated":"2026-05-26T04:07:55Z","snapshot_observed_at":"2026-08-06T03:28:38.522756Z","submitted_at":"2026-05-26T04:07:55Z","title":"Testing Agentic Workflows with Structural Coverage Criteria","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-06-29T16:17:28.497424Z"},"links":{"cited_paper":"/paper/2304.08244","citing_paper":"/paper/2605.26521"},"observation_digest":"sha256:a86d4d0ce4c8ce40780c46ea84fc2b847008a39c4f085f5d0af57c12a8ad08bd","observation_id":"462c8bb5-f17f-4410-b4ef-16686b16daa5","resolution":{"observed_at":"2026-06-29T16:23:39.597917Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-07-14T18:20:23.914543+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-14T18:20:23.914543+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.08244","last_updated":"2023-10-25T06:54:12Z","snapshot_observed_at":"2026-08-02T00:07:12.855748Z","submitted_at":"2023-04-14T14:05:32Z","title":"API-Bank: A Comprehensive Benchmark for Tool-Augmented LLMs","version":2},"cited_work":{"arxiv_id":"2304.08244","doi":"10.48550/arxiv.2304.08244","metadata_source":"pith","pith_arxiv_id":"2304.08244","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"API-Bank: A Comprehensive Benchmark for Tool-Augmented LLMs","venue":"cs.CL","work_id":"a20d9332-ab34-485c-a060-1ba47cc98930","year":2023},"citing_paper":{"arxiv_id":"2605.31478","last_updated":"2026-05-29T16:06:34Z","snapshot_observed_at":"2026-07-06T23:40:36.926513Z","submitted_at":"2026-05-29T16:06:34Z","title":"Knowledge Boundary Probing and Demand-Guided Intervention for LLM-Based Power System Code Generation","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-06-28T21:25:51.439330Z"},"links":{"cited_paper":"/paper/2304.08244","citing_paper":"/paper/2605.31478"},"observation_digest":"sha256:be3f81b7aa4f5b771b6a226e307e38cf975f5cac3d0104d41cd9023d16dccf30","observation_id":"1d3c429f-c7ef-4adb-97e0-a6f671ca604c","resolution":{"observed_at":"2026-07-01T20:16:11.834422Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-07-14T18:20:23.914543+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-14T18:20:23.914543+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.08244","last_updated":"2023-10-25T06:54:12Z","snapshot_observed_at":"2026-08-02T00:07:12.855748Z","submitted_at":"2023-04-14T14:05:32Z","title":"API-Bank: A Comprehensive Benchmark for Tool-Augmented LLMs","version":2},"cited_work":{"arxiv_id":"2304.08244","doi":"10.48550/arxiv.2304.08244","metadata_source":"pith","pith_arxiv_id":"2304.08244","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"API-Bank: A Comprehensive Benchmark for Tool-Augmented LLMs","venue":"cs.CL","work_id":"a20d9332-ab34-485c-a060-1ba47cc98930","year":2023},"citing_paper":{"arxiv_id":"2606.03854","last_updated":"2026-06-02T16:30:33Z","snapshot_observed_at":"2026-08-06T14:43:37.104974Z","submitted_at":"2026-06-02T16:30:33Z","title":"CLI-Anything: Towards Agent-Native Computer Use","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-06-28T08:20:03.855573Z"},"links":{"cited_paper":"/paper/2304.08244","citing_paper":"/paper/2606.03854"},"observation_digest":"sha256:c0f217bc4604cd0ad48eb57872f18433c645dbcfdbe4c5a5f1581e014e29a19c","observation_id":"919b48d2-47cd-4d29-91fc-a3a499af0a07","resolution":{"observed_at":"2026-07-02T05:16:39.588710Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-07-14T18:20:23.914543+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-14T18:20:23.914543+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.08244","last_updated":"2023-10-25T06:54:12Z","snapshot_observed_at":"2026-08-02T00:07:12.855748Z","submitted_at":"2023-04-14T14:05:32Z","title":"API-Bank: A Comprehensive Benchmark for Tool-Augmented LLMs","version":2},"cited_work":{"arxiv_id":"2304.08244","doi":"10.48550/arxiv.2304.08244","metadata_source":"pith","pith_arxiv_id":"2304.08244","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"API-Bank: A Comprehensive Benchmark for Tool-Augmented LLMs","venue":"cs.CL","work_id":"a20d9332-ab34-485c-a060-1ba47cc98930","year":2023},"citing_paper":{"arxiv_id":"2606.06566","last_updated":"2026-06-04T17:27:39Z","snapshot_observed_at":"2026-08-06T00:01:29.832904Z","submitted_at":"2026-06-04T17:27:39Z","title":"NTILC: Neural Tool Invocation via Learned Compression","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-06-28T00:05:36.600389Z"},"links":{"cited_paper":"/paper/2304.08244","citing_paper":"/paper/2606.06566"},"observation_digest":"sha256:12f2b463daa7a118b7206429cdedb0af9af05c9ba6f90e3349caac0460caac73","observation_id":"f603fe7d-7fbd-48ac-811a-c1f17e964cf5","resolution":{"observed_at":"2026-07-02T15:07:04.802789Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-07-14T18:20:23.914543+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-14T18:20:23.914543+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.08244","last_updated":"2023-10-25T06:54:12Z","snapshot_observed_at":"2026-08-02T00:07:12.855748Z","submitted_at":"2023-04-14T14:05:32Z","title":"API-Bank: A Comprehensive Benchmark for Tool-Augmented LLMs","version":2},"cited_work":{"arxiv_id":"2304.08244","doi":"10.48550/arxiv.2304.08244","metadata_source":"pith","pith_arxiv_id":"2304.08244","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"API-Bank: A Comprehensive Benchmark for Tool-Augmented LLMs","venue":"cs.CL","work_id":"a20d9332-ab34-485c-a060-1ba47cc98930","year":2023},"citing_paper":{"arxiv_id":"2606.07904","last_updated":"2026-06-05T23:47:33Z","snapshot_observed_at":"2026-08-01T22:52:28.866876Z","submitted_at":"2026-06-05T23:47:33Z","title":"Contract2Tool: Learning Preconditions and Effects for Reliable Tool-Augmented LLM Agents","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-06-27T21:32:06.237095Z"},"links":{"cited_paper":"/paper/2304.08244","citing_paper":"/paper/2606.07904"},"observation_digest":"sha256:8d32ed410fcb23268f84b897819ef71b19a0208ebbf422150cf26b9d69eb9d3c","observation_id":"20807d09-fb00-4ec9-b8cc-6b686d58eee2","resolution":{"observed_at":"2026-07-02T19:17:18.709187Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-07-14T18:20:23.914543+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-14T18:20:23.914543+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.08244","last_updated":"2023-10-25T06:54:12Z","snapshot_observed_at":"2026-08-02T00:07:12.855748Z","submitted_at":"2023-04-14T14:05:32Z","title":"API-Bank: A Comprehensive Benchmark for Tool-Augmented LLMs","version":2},"cited_work":{"arxiv_id":"2304.08244","doi":"10.48550/arxiv.2304.08244","metadata_source":"pith","pith_arxiv_id":"2304.08244","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"API-Bank: A Comprehensive Benchmark for Tool-Augmented LLMs","venue":"cs.CL","work_id":"a20d9332-ab34-485c-a060-1ba47cc98930","year":2023},"citing_paper":{"arxiv_id":"2606.10106","last_updated":"2026-06-08T19:35:37Z","snapshot_observed_at":"2026-08-01T08:26:23.637148Z","submitted_at":"2026-06-08T19:35:37Z","title":"What makes a harness a harness: necessary and sufficient conditions for an agent harness","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-06-27T15:15:57.372858Z"},"links":{"cited_paper":"/paper/2304.08244","citing_paper":"/paper/2606.10106"},"observation_digest":"sha256:92e59b0943245cb8d82be695d46a133a3773a65f5b23106c01ba855686a1b461","observation_id":"8d9934ab-f58c-49e5-bc0a-366ad3d090b8","resolution":{"observed_at":"2026-06-27T19:31:10.865229Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-07-14T18:20:23.914543+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-14T18:20:23.914543+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.08244","last_updated":"2023-10-25T06:54:12Z","snapshot_observed_at":"2026-08-02T00:07:12.855748Z","submitted_at":"2023-04-14T14:05:32Z","title":"API-Bank: A Comprehensive Benchmark for Tool-Augmented LLMs","version":2},"cited_work":{"arxiv_id":"2304.08244","doi":"10.48550/arxiv.2304.08244","metadata_source":"pith","pith_arxiv_id":"2304.08244","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"API-Bank: A Comprehensive Benchmark for Tool-Augmented LLMs","venue":"cs.CL","work_id":"a20d9332-ab34-485c-a060-1ba47cc98930","year":2023},"citing_paper":{"arxiv_id":"2606.11830","last_updated":"2026-06-10T09:13:10Z","snapshot_observed_at":"2026-08-07T15:34:57.329668Z","submitted_at":"2026-06-10T09:13:10Z","title":"Skill-Augmented AI Agents for Medical Research Analysis: An Exploratory Multi-Model Human Evaluation in an NSCLC Transcriptomic Biomarker Task","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-06-27T09:55:01.205573Z"},"links":{"cited_paper":"/paper/2304.08244","citing_paper":"/paper/2606.11830"},"observation_digest":"sha256:0d3d4bba21ab55a89fcd3f40046ecc27a363b8e3e371028bf1b7fdfcdcb68dcd","observation_id":"b138915d-492a-4264-ab6d-6a0b7eadd3e1","resolution":{"observed_at":"2026-07-03T10:37:56.932757Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-07-14T18:20:23.914543+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-14T18:20:23.914543+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.08244","last_updated":"2023-10-25T06:54:12Z","snapshot_observed_at":"2026-08-02T00:07:12.855748Z","submitted_at":"2023-04-14T14:05:32Z","title":"API-Bank: A Comprehensive Benchmark for Tool-Augmented LLMs","version":2},"cited_work":{"arxiv_id":"2304.08244","doi":"10.48550/arxiv.2304.08244","metadata_source":"pith","pith_arxiv_id":"2304.08244","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"API-Bank: A Comprehensive Benchmark for Tool-Augmented LLMs","venue":"cs.CL","work_id":"a20d9332-ab34-485c-a060-1ba47cc98930","year":2023},"citing_paper":{"arxiv_id":"2606.12908","last_updated":"2026-06-11T05:06:50Z","snapshot_observed_at":"2026-08-05T04:59:46.668803Z","submitted_at":"2026-06-11T05:06:50Z","title":"SENTINEL: Failure-Driven Reinforcement Learning for Training Tool-Using Language Model Agents","version":1},"reference_index":15,"source":"arxiv_source","source_observed_at":"2026-06-27T06:49:12.070481Z"},"links":{"cited_paper":"/paper/2304.08244","citing_paper":"/paper/2606.12908"},"observation_digest":"sha256:e3ca5f3db579923e69f243c1591e2759378f473196ec50c578aac8fa025f5c52","observation_id":"c879f485-c2c6-4d80-8a1b-9e02d4ce5885","resolution":{"observed_at":"2026-07-03T14:58:33.242010Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-07-14T18:20:23.914543+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-14T18:20:23.914543+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.08244","last_updated":"2023-10-25T06:54:12Z","snapshot_observed_at":"2026-08-02T00:07:12.855748Z","submitted_at":"2023-04-14T14:05:32Z","title":"API-Bank: A Comprehensive Benchmark for Tool-Augmented LLMs","version":2},"cited_work":{"arxiv_id":"2304.08244","doi":"10.48550/arxiv.2304.08244","metadata_source":"pith","pith_arxiv_id":"2304.08244","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"API-Bank: A Comprehensive Benchmark for Tool-Augmented LLMs","venue":"cs.CL","work_id":"a20d9332-ab34-485c-a060-1ba47cc98930","year":2023},"citing_paper":{"arxiv_id":"2606.17162","last_updated":"2026-06-15T18:02:55Z","snapshot_observed_at":"2026-08-01T11:32:11.709711Z","submitted_at":"2026-06-15T18:02:55Z","title":"MemSlides: A Hierarchical Memory Driven Agent Framework for Personalized Slide Generation with Multi-turn Local Revision","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-06-27T03:43:37.671325Z"},"links":{"cited_paper":"/paper/2304.08244","citing_paper":"/paper/2606.17162"},"observation_digest":"sha256:34d4d8310d662e1048a8ad1921d5068d9e28668e859bee4536b7e5622e3c76d3","observation_id":"36a52fef-6599-4810-afd2-29e4bad1929e","resolution":{"observed_at":"2026-07-03T17:48:45.994355Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-07-14T18:20:23.914543+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-14T18:20:23.914543+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.08244","last_updated":"2023-10-25T06:54:12Z","snapshot_observed_at":"2026-08-02T00:07:12.855748Z","submitted_at":"2023-04-14T14:05:32Z","title":"API-Bank: A Comprehensive Benchmark for Tool-Augmented LLMs","version":2},"cited_work":{"arxiv_id":"2304.08244","doi":"10.48550/arxiv.2304.08244","metadata_source":"pith","pith_arxiv_id":"2304.08244","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"API-Bank: A Comprehensive Benchmark for Tool-Augmented LLMs","venue":"cs.CL","work_id":"a20d9332-ab34-485c-a060-1ba47cc98930","year":2023},"citing_paper":{"arxiv_id":"2606.19409","last_updated":"2026-06-17T15:19:19Z","snapshot_observed_at":"2026-08-03T01:26:10.680902Z","submitted_at":"2026-06-17T15:19:19Z","title":"OpenRath: Session-Centered Runtime State for Agent Systems","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-06-26T20:18:48.243426Z"},"links":{"cited_paper":"/paper/2304.08244","citing_paper":"/paper/2606.19409"},"observation_digest":"sha256:2933602c1ac62f9c3698a6a888d81e91bf6345700048f865c4e3274f67d26ec4","observation_id":"15941e6f-3eb8-4382-9ea3-2ff3da0c28cb","resolution":{"observed_at":"2026-07-04T01:29:23.012616Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-07-14T18:20:23.914543+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-14T18:20:23.914543+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.08244","last_updated":"2023-10-25T06:54:12Z","snapshot_observed_at":"2026-08-02T00:07:12.855748Z","submitted_at":"2023-04-14T14:05:32Z","title":"API-Bank: A Comprehensive Benchmark for Tool-Augmented LLMs","version":2},"cited_work":{"arxiv_id":"2304.08244","doi":"10.48550/arxiv.2304.08244","metadata_source":"pith","pith_arxiv_id":"2304.08244","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"API-Bank: A Comprehensive Benchmark for Tool-Augmented LLMs","venue":"cs.CL","work_id":"a20d9332-ab34-485c-a060-1ba47cc98930","year":2023},"citing_paper":{"arxiv_id":"2606.23049","last_updated":"2026-06-24T02:20:35Z","snapshot_observed_at":"2026-07-06T23:57:53.101659Z","submitted_at":"2026-06-22T08:57:54Z","title":"PhoneBuddy: Training Open Models for Agentic Phone Use","version":2},"reference_index":36,"source":"arxiv_source","source_observed_at":"2026-06-26T08:15:49.428124Z"},"links":{"cited_paper":"/paper/2304.08244","citing_paper":"/paper/2606.23049"},"observation_digest":"sha256:9cc2fe491ae82f300c89459a48b2d180ca2b684c26c89a7d1a132a49c75bcb56","observation_id":"81ef9cf6-3008-4e06-8613-639d8eca84c9","resolution":{"observed_at":"2026-07-04T10:59:46.469016Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-07-14T18:20:23.914543+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-14T18:20:23.914543+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.08244","last_updated":"2023-10-25T06:54:12Z","snapshot_observed_at":"2026-08-02T00:07:12.855748Z","submitted_at":"2023-04-14T14:05:32Z","title":"API-Bank: A Comprehensive Benchmark for Tool-Augmented LLMs","version":2},"cited_work":{"arxiv_id":"2304.08244","doi":"10.48550/arxiv.2304.08244","metadata_source":"pith","pith_arxiv_id":"2304.08244","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"API-Bank: A Comprehensive Benchmark for Tool-Augmented LLMs","venue":"cs.CL","work_id":"a20d9332-ab34-485c-a060-1ba47cc98930","year":2023},"citing_paper":{"arxiv_id":"2606.28715","last_updated":"2026-06-27T03:44:00Z","snapshot_observed_at":"2026-08-07T02:52:16.765972Z","submitted_at":"2026-06-27T03:44:00Z","title":"SEATauBench: Adapting Tool-Agent-User Evaluation Into Low-Resource Southeast Asian Languages","version":1},"reference_index":10,"source":"arxiv_source","source_observed_at":"2026-06-30T10:12:45.090257Z"},"links":{"cited_paper":"/paper/2304.08244","citing_paper":"/paper/2606.28715"},"observation_digest":"sha256:cae23d7c7cb028a764a0f2eadf602556c02e8e3f98f2f8655aa7ed5e7c45bcc2","observation_id":"479681bb-ff1f-4019-9b38-40818e53bcda","resolution":{"observed_at":"2026-06-30T10:14:36.201722Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-07-14T18:20:23.914543+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-14T18:20:23.914543+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.08244","last_updated":"2023-10-25T06:54:12Z","snapshot_observed_at":"2026-08-02T00:07:12.855748Z","submitted_at":"2023-04-14T14:05:32Z","title":"API-Bank: A Comprehensive Benchmark for Tool-Augmented LLMs","version":2},"cited_work":{"arxiv_id":"2304.08244","doi":"10.48550/arxiv.2304.08244","metadata_source":"pith","pith_arxiv_id":"2304.08244","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"API-Bank: A Comprehensive Benchmark for Tool-Augmented LLMs","venue":"cs.CL","work_id":"a20d9332-ab34-485c-a060-1ba47cc98930","year":2023},"citing_paper":{"arxiv_id":"2606.30775","last_updated":"2026-06-29T18:06:43Z","snapshot_observed_at":"2026-07-07T00:04:34.491408Z","submitted_at":"2026-06-29T18:06:43Z","title":"A Single Rewrite Suffices: Empirical Lessons from Production Skill Description Optimization","version":1},"reference_index":175,"source":"arxiv_source","source_observed_at":"2026-07-01T02:32:19.425550Z"},"links":{"cited_paper":"/paper/2304.08244","citing_paper":"/paper/2606.30775"},"observation_digest":"sha256:ecede6df6876918a2e27948417a6f8e9ba491818d18d21c13fe9e4deeb50e3d2","observation_id":"a8d4f677-7147-4af1-88ec-f9d958d93e3f","resolution":{"observed_at":"2026-07-01T12:05:43.578203Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-07-14T18:20:23.914543+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-14T18:20:23.914543+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.08244","last_updated":"2023-10-25T06:54:12Z","snapshot_observed_at":"2026-08-02T00:07:12.855748Z","submitted_at":"2023-04-14T14:05:32Z","title":"API-Bank: A Comprehensive Benchmark for Tool-Augmented LLMs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.08244","snapshot_observed_at":"2026-07-12T09:42:05.329537Z","title":"Api-bank: A comprehensive benchmark for tool-augmented llms.arXiv preprint arXiv:2304.08244, 2023a","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.02577","last_updated":"2026-06-30T21:10:42Z","snapshot_observed_at":"2026-08-07T06:54:53.060230Z","submitted_at":"2026-06-30T21:10:42Z","title":"Benchmarking the Benchmarks: A Validity Audit of Tool-Calling Evaluation","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-07-12T09:42:05.329537Z"},"links":{"cited_paper":"/paper/2304.08244","citing_paper":"/paper/2607.02577"},"observation_digest":"sha256:f1b056022a12302a20e76b711b9587fbfd0e2b0496256f995103a8b16760d2fd","observation_id":"ba1ca775-7917-43df-b323-7d70bdee1b3b","resolution":{"observed_at":"2026-07-12T09:42:05.329537Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.08244","last_updated":"2023-10-25T06:54:12Z","snapshot_observed_at":"2026-08-02T00:07:12.855748Z","submitted_at":"2023-04-14T14:05:32Z","title":"API-Bank: A Comprehensive Benchmark for Tool-Augmented LLMs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.08244","snapshot_observed_at":"2026-07-13T05:59:23.238387Z","title":"arXiv preprint arXiv:2304.08244 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.08894","last_updated":"2026-07-09T19:34:29Z","snapshot_observed_at":"2026-07-15T23:17:56.360081Z","submitted_at":"2026-07-09T19:34:29Z","title":"GATS: Graph-Augmented Tree Search with Layered World Models for Efficient Agent Planning","version":1},"reference_index":20,"source":"arxiv_source","source_observed_at":"2026-07-13T05:59:23.238387Z"},"links":{"cited_paper":"/paper/2304.08244","citing_paper":"/paper/2607.08894"},"observation_digest":"sha256:d5e3f0971d931513d9908dd58a9b7c97075ee2b22a59af11bde11fb1b52dfa27","observation_id":"d16bf914-0b5c-4ffd-9d89-4c313ed8f701","resolution":{"observed_at":"2026-07-13T05:59:23.238387Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.08244","last_updated":"2023-10-25T06:54:12Z","snapshot_observed_at":"2026-08-02T00:07:12.855748Z","submitted_at":"2023-04-14T14:05:32Z","title":"API-Bank: A Comprehensive Benchmark for Tool-Augmented LLMs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.08244","snapshot_observed_at":"2026-07-13T03:29:34.486347Z","title":"Api-bank: A comprehensive benchmark for tool-augmented llms.arXiv preprint arXiv:2304.08244, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.09375","last_updated":"2026-07-10T12:57:08Z","snapshot_observed_at":"2026-08-07T07:07:24.560093Z","submitted_at":"2026-07-10T12:57:08Z","title":"Mach-Mind-4-Flash Technical Report","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-07-13T03:29:34.486347Z"},"links":{"cited_paper":"/paper/2304.08244","citing_paper":"/paper/2607.09375"},"observation_digest":"sha256:dadfc0caf6a3436939615194bcab574d310420b496908d7cffa594e8273b4885","observation_id":"491e6537-7e74-40ae-8b6f-40011538cbab","resolution":{"observed_at":"2026-07-13T03:29:34.486347Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.08244","last_updated":"2023-10-25T06:54:12Z","snapshot_observed_at":"2026-08-02T00:07:12.855748Z","submitted_at":"2023-04-14T14:05:32Z","title":"API-Bank: A Comprehensive Benchmark for Tool-Augmented LLMs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.08244","snapshot_observed_at":"2026-07-13T02:40:34.279112Z","title":"API-Bank: A benchmark for tool-augmented LLMs, 2023.https://arxiv.org/abs/2304.08244","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.09489","last_updated":"2026-07-10T15:05:28Z","snapshot_observed_at":"2026-08-02T23:34:28.338701Z","submitted_at":"2026-07-10T15:05:28Z","title":"Ceci n'est pas une pipe: AI systems as semantic abstractions","version":1},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-07-13T02:40:34.279112Z"},"links":{"cited_paper":"/paper/2304.08244","citing_paper":"/paper/2607.09489"},"observation_digest":"sha256:186621ad9f2b76ba2e6370329b70991bbe60188beda0bf7d8dfb21a72c582cf8","observation_id":"1db43143-b52d-451f-9bf6-74d1f40a6b1d","resolution":{"observed_at":"2026-07-13T02:40:34.279112Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.08244","last_updated":"2023-10-25T06:54:12Z","snapshot_observed_at":"2026-08-02T00:07:12.855748Z","submitted_at":"2023-04-14T14:05:32Z","title":"API-Bank: A Comprehensive Benchmark for Tool-Augmented LLMs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.08244","snapshot_observed_at":"2026-07-14T08:37:17.973015Z","title":"2023 , howpublished =","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.10878","last_updated":"2026-07-12T18:46:42Z","snapshot_observed_at":"2026-08-04T09:27:03.135071Z","submitted_at":"2026-07-12T18:46:42Z","title":"LOGOS: A Living Logic for AI Agent Teams That Evolve With Humans","version":1},"reference_index":43,"source":"arxiv_source","source_observed_at":"2026-07-14T08:37:17.973015Z"},"links":{"cited_paper":"/paper/2304.08244","citing_paper":"/paper/2607.10878"},"observation_digest":"sha256:1ab8b1a675bf85681d1f4c853d98697b2a9f65134cb8937263a3822e09565046","observation_id":"9ea1bd16-bafd-4775-a936-3167f14c6c39","resolution":{"observed_at":"2026-07-14T08:37:17.973015Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.08244","last_updated":"2023-10-25T06:54:12Z","snapshot_observed_at":"2026-08-02T00:07:12.855748Z","submitted_at":"2023-04-14T14:05:32Z","title":"API-Bank: A Comprehensive Benchmark for Tool-Augmented LLMs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.08244","snapshot_observed_at":"2026-08-01T21:29:41.718592Z","title":"API-Bank: A comprehensive benchmark for tool-augmented LLMs.arXiv preprint arXiv:2304.08244, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.16066","last_updated":"2026-07-17T15:47:23Z","snapshot_observed_at":"2026-08-01T21:29:31.125079Z","submitted_at":"2026-07-17T15:47:23Z","title":"LLM-Powered Agentic AI for 5G/6G Networks: A Tutorial and Survey on Architectures, Protocols, and Standardization","version":1},"reference_index":65,"source":"pdf_text","source_observed_at":"2026-08-01T21:29:41.718592Z"},"links":{"cited_paper":"/paper/2304.08244","citing_paper":"/paper/2607.16066"},"observation_digest":"sha256:7df1eb67f7c5cfd11363fc70fcb4f703259cb94a2d66be9352d848b2fd35f170","observation_id":"0d4518c6-0ccc-4314-a513-d460d619016c","resolution":{"observed_at":"2026-08-01T21:29:41.718592Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.08244","last_updated":"2023-10-25T06:54:12Z","snapshot_observed_at":"2026-08-02T00:07:12.855748Z","submitted_at":"2023-04-14T14:05:32Z","title":"API-Bank: A Comprehensive Benchmark for Tool-Augmented LLMs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.08244","snapshot_observed_at":"2026-08-01T17:18:05.283498Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.17701","last_updated":"2026-07-20T08:48:36Z","snapshot_observed_at":"2026-08-06T04:38:11.453039Z","submitted_at":"2026-07-20T08:48:36Z","title":"ProEvent: An Event-centric Benchmark for Proactive Agents","version":1},"reference_index":52,"source":"arxiv_source","source_observed_at":"2026-08-01T17:18:05.283498Z"},"links":{"cited_paper":"/paper/2304.08244","citing_paper":"/paper/2607.17701"},"observation_digest":"sha256:81f5ae94ea975afebc4f4a28ed41a200a2aad0fd7f420dffec5f969c64aa53db","observation_id":"f4806c61-d969-4397-857a-c0651b09f023","resolution":{"observed_at":"2026-08-01T17:18:05.283498Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.08244","last_updated":"2023-10-25T06:54:12Z","snapshot_observed_at":"2026-08-02T00:07:12.855748Z","submitted_at":"2023-04-14T14:05:32Z","title":"API-Bank: A Comprehensive Benchmark for Tool-Augmented LLMs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.08244","snapshot_observed_at":"2026-08-01T15:29:25.830028Z","title":"API-Bank: A Comprehensive Benchmark for Tool-Augmented LLMs","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.18438","last_updated":"2026-07-20T18:46:17Z","snapshot_observed_at":"2026-08-07T06:34:55.886504Z","submitted_at":"2026-07-20T18:46:17Z","title":"Relay-Bench: Evaluating LLMs on Multi-Domain Reasoning Chains","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-01T15:29:25.830028Z"},"links":{"cited_paper":"/paper/2304.08244","citing_paper":"/paper/2607.18438"},"observation_digest":"sha256:264f5b16471950da46b349bc1c97c44833124a61842a7ae5b19831e8cdfa6775","observation_id":"3e48715e-bf87-4070-8042-f448aa32867b","resolution":{"observed_at":"2026-08-01T15:29:25.830028Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.08244","last_updated":"2023-10-25T06:54:12Z","snapshot_observed_at":"2026-08-02T00:07:12.855748Z","submitted_at":"2023-04-14T14:05:32Z","title":"API-Bank: A Comprehensive Benchmark for Tool-Augmented LLMs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.08244","snapshot_observed_at":"2026-08-01T15:03:42.730768Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.21635","last_updated":"2026-07-20T23:14:36Z","snapshot_observed_at":"2026-08-05T00:41:56.304371Z","submitted_at":"2026-07-20T23:14:36Z","title":"Toward User-Conditioned Evaluation of Personal LLM Agents under Temporal Interventions","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-01T15:03:42.730768Z"},"links":{"cited_paper":"/paper/2304.08244","citing_paper":"/paper/2607.21635"},"observation_digest":"sha256:0602f4de943601bbb61cfb1bac62fc8e0def6a153300973c534dc4ec6f2d65b7","observation_id":"abd8a518-7536-49d2-bcca-e703f9fe52ed","resolution":{"observed_at":"2026-08-01T15:03:42.730768Z","resolver_source":null,"status":"malformed_identifier"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.08244","last_updated":"2023-10-25T06:54:12Z","snapshot_observed_at":"2026-08-02T00:07:12.855748Z","submitted_at":"2023-04-14T14:05:32Z","title":"API-Bank: A Comprehensive Benchmark for Tool-Augmented LLMs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.08244","snapshot_observed_at":"2026-08-02T10:18:37.495334Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.22642","last_updated":"2026-06-24T00:12:03Z","snapshot_observed_at":"2026-08-07T04:06:13.678510Z","submitted_at":"2026-06-24T00:12:03Z","title":"CRAFT: Learn the Schema, Execute the Plan","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-02T10:18:37.495334Z"},"links":{"cited_paper":"/paper/2304.08244","citing_paper":"/paper/2607.22642"},"observation_digest":"sha256:3e8e429dfc285168b163592dd1bd60272d7680e04c1144a3977e0f03ca2a2055","observation_id":"e233c450-2dbf-48af-ba42-cade0ca1db5a","resolution":{"observed_at":"2026-08-02T10:18:37.495334Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.08244","last_updated":"2023-10-25T06:54:12Z","snapshot_observed_at":"2026-08-02T00:07:12.855748Z","submitted_at":"2023-04-14T14:05:32Z","title":"API-Bank: A Comprehensive Benchmark for Tool-Augmented LLMs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.08244","snapshot_observed_at":"2026-08-01T03:35:43.031171Z","title":"API-Bank: A comprehensive benchmark for tool-augmented LLMs.arXiv preprint arXiv:2304.08244, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.23123","last_updated":"2026-07-25T09:55:31Z","snapshot_observed_at":"2026-08-07T08:46:33.575663Z","submitted_at":"2026-07-25T09:55:31Z","title":"SQBench: A Benchmark for Evaluating Task Delivery by Language-Model Agents in Production-Oriented Workflows","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-01T03:35:43.031171Z"},"links":{"cited_paper":"/paper/2304.08244","citing_paper":"/paper/2607.23123"},"observation_digest":"sha256:704c4588870207296704c1ba04bb57deaee298512d703917685943f39c232230","observation_id":"d0dda371-ab98-4916-b787-30621babd5a5","resolution":{"observed_at":"2026-08-01T03:35:43.031171Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.08244","last_updated":"2023-10-25T06:54:12Z","snapshot_observed_at":"2026-08-02T00:07:12.855748Z","submitted_at":"2023-04-14T14:05:32Z","title":"API-Bank: A Comprehensive Benchmark for Tool-Augmented LLMs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.08244","snapshot_observed_at":"2026-08-01T03:38:14.153015Z","title":"API-Bank: A comprehensive benchmark for tool-augmented LLMs.arXiv preprint arXiv:2304.08244, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.23124","last_updated":"2026-07-25T09:58:24Z","snapshot_observed_at":"2026-08-06T19:50:20.416891Z","submitted_at":"2026-07-25T09:58:24Z","title":"AgentOmnia: Scaling Agentic Models for Full-Scenario Applications","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-01T03:38:14.153015Z"},"links":{"cited_paper":"/paper/2304.08244","citing_paper":"/paper/2607.23124"},"observation_digest":"sha256:fb2fa4bc0264da301dfa568e67c03872bb2078c87c73331e1f818618ab224731","observation_id":"ceb3b68c-bc60-4a3a-9cb4-8172151e9d01","resolution":{"observed_at":"2026-08-01T03:38:14.153015Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.08244","last_updated":"2023-10-25T06:54:12Z","snapshot_observed_at":"2026-08-02T00:07:12.855748Z","submitted_at":"2023-04-14T14:05:32Z","title":"API-Bank: A Comprehensive Benchmark for Tool-Augmented LLMs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.08244","snapshot_observed_at":"2026-08-01T01:35:09.670186Z","title":"Api-bank: A comprehensive benchmark for tool-augmented llms","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.25765","last_updated":"2026-07-28T14:19:59Z","snapshot_observed_at":"2026-08-07T07:35:13.308973Z","submitted_at":"2026-07-28T14:19:59Z","title":"WorkSurface-Bench: Benchmarking Enterprise Agents on Multi-Surface Knowledge Routing","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-01T01:35:09.670186Z"},"links":{"cited_paper":"/paper/2304.08244","citing_paper":"/paper/2607.25765"},"observation_digest":"sha256:dbe72484378088592e08e34d9af808e04fa107c8b111539b64db5a23f020d1fc","observation_id":"f53554aa-2c8a-4243-a619-d863f7379586","resolution":{"observed_at":"2026-08-01T01:35:09.670186Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.08244","last_updated":"2023-10-25T06:54:12Z","snapshot_observed_at":"2026-08-02T00:07:12.855748Z","submitted_at":"2023-04-14T14:05:32Z","title":"API-Bank: A Comprehensive Benchmark for Tool-Augmented LLMs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.08244","snapshot_observed_at":"2026-08-03T12:17:55.456005Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.29175","last_updated":"2026-07-31T08:52:26Z","snapshot_observed_at":"2026-08-06T17:19:53.625782Z","submitted_at":"2026-07-31T08:52:26Z","title":"Execution-First Synthetic Tool-Use Trace Generation for LLM Agents","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-03T12:17:55.456005Z"},"links":{"cited_paper":"/paper/2304.08244","citing_paper":"/paper/2607.29175"},"observation_digest":"sha256:8f778a37857aa4d6fbb8aa77ecd786a019097f1bdcc94478126564bede7392aa","observation_id":"18030774-6878-4337-9381-1420c505e1b7","resolution":{"observed_at":"2026-08-03T12:17:55.456005Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.08244","last_updated":"2023-10-25T06:54:12Z","snapshot_observed_at":"2026-08-02T00:07:12.855748Z","submitted_at":"2023-04-14T14:05:32Z","title":"API-Bank: A Comprehensive Benchmark for Tool-Augmented LLMs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.08244","snapshot_observed_at":"2026-08-06T04:46:38.571002Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2608.05126","last_updated":"2026-08-05T17:50:31Z","snapshot_observed_at":"2026-08-07T15:14:24.148112Z","submitted_at":"2026-08-05T17:50:31Z","title":"Spoken Function Calling: A New Perspective on Spoken Language Understanding for Large Audio Language Models","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-06T04:46:38.571002Z"},"links":{"cited_paper":"/paper/2304.08244","citing_paper":"/paper/2608.05126"},"observation_digest":"sha256:e6b9d408ffda25011119935de7510ef8a70be0920223025a98ca63e2eab74ecc","observation_id":"df2f28b3-d29a-4b28-a0b7-389c682f1a60","resolution":{"observed_at":"2026-08-06T04:46:38.571002Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2304.08244/citation-record","integrity":"/paper/2304.08244/integrity","json":"/paper/2304.08244/citation-record.json","paper":"/paper/2304.08244"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Advances in neural information processing systems, 33:1877–1901","venue":null,"work_id":"91595e80-9195-4fc9-8ee1-b418a4f1eb6c","year":1901},"citing_paper":{"arxiv_id":"2304.08244","last_updated":"2023-10-25T06:54:12Z","snapshot_observed_at":"2026-08-02T00:07:12.855748Z","submitted_at":"2023-04-14T14:05:32Z","title":"API-Bank: A Comprehensive Benchmark for Tool-Augmented LLMs","version":2},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-15T20:51:41.198767Z"},"links":{"citing_paper":"/paper/2304.08244"},"observation_digest":"sha256:b3413939af3337d509a91bffa13bf643f69f9a0a1ea9c0781f4595bf7dd3a547","observation_id":"7e8c62eb-e041-47e0-8dcf-1a388158b592","resolution":{"observed_at":"2026-05-15T20:51:41.274960Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2303.12712","last_updated":"2023-04-13T20:41:31Z","snapshot_observed_at":"2026-08-03T04:49:15.195814Z","submitted_at":"2023-03-22T16:51:28Z","title":"Sparks of Artificial General Intelligence: Early experiments with GPT-4","version":5},"cited_work":{"arxiv_id":"2303.12712","doi":"10.48550/arxiv.2303.12712","metadata_source":"pith","pith_arxiv_id":"2303.12712","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Sparks of Artificial General Intelligence: Early experiments with GPT-4","venue":"cs.CL","work_id":"a23cfe92-7f7c-424b-98d4-b386a83002fb","year":2023},"citing_paper":{"arxiv_id":"2304.08244","last_updated":"2023-10-25T06:54:12Z","snapshot_observed_at":"2026-08-02T00:07:12.855748Z","submitted_at":"2023-04-14T14:05:32Z","title":"API-Bank: A Comprehensive Benchmark for Tool-Augmented LLMs","version":2},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-15T20:51:41.198767Z"},"links":{"cited_paper":"/paper/2303.12712","citing_paper":"/paper/2304.08244"},"observation_digest":"sha256:95551ffec217229ace2201f49f74e3d7175ddc72ee2fd92648707d6635ef8648","observation_id":"30f8b389-5a5a-42b2-89b7-5f88fa2c3b2f","resolution":{"observed_at":"2026-05-15T20:51:41.214006Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.17126","last_updated":"2024-03-11T01:15:09Z","snapshot_observed_at":"2026-08-05T01:49:57.023215Z","submitted_at":"2023-05-26T17:50:11Z","title":"Large Language Models as Tool Makers","version":2},"cited_work":{"arxiv_id":"2305.17126","doi":"10.48550/arxiv.2305.17126","metadata_source":"arxiv_reference","pith_arxiv_id":"2305.17126","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Large language models as tool makers","venue":"arXiv (Cornell University)","work_id":"1447b78e-0a79-4af6-8cd4-93220e680d2b","year":2023},"citing_paper":{"arxiv_id":"2304.08244","last_updated":"2023-10-25T06:54:12Z","snapshot_observed_at":"2026-08-02T00:07:12.855748Z","submitted_at":"2023-04-14T14:05:32Z","title":"API-Bank: A Comprehensive Benchmark for Tool-Augmented LLMs","version":2},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-15T20:51:41.198767Z"},"links":{"cited_paper":"/paper/2305.17126","citing_paper":"/paper/2304.08244"},"observation_digest":"sha256:9fa04ec8785d2c9a500e6e58d3457dd97a7d06782ee18736cd383deab11e9df5","observation_id":"eb4fc248-f863-459b-bfd1-6b4f69cc96e1","resolution":{"observed_at":"2026-05-15T20:51:41.217174Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2107.03374","last_updated":"2021-07-14T17:16:02Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2021-07-07T17:41:24Z","title":"Evaluating Large Language Models Trained on Code","version":2},"cited_work":{"arxiv_id":"2107.03374","doi":"10.48550/arxiv.2107.03374","metadata_source":"pith","pith_arxiv_id":"2107.03374","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Evaluating Large Language Models Trained on Code","venue":"cs.LG","work_id":"042493e9-b26f-4b4e-bbde-382072ca9b08","year":2021},"citing_paper":{"arxiv_id":"2304.08244","last_updated":"2023-10-25T06:54:12Z","snapshot_observed_at":"2026-08-02T00:07:12.855748Z","submitted_at":"2023-04-14T14:05:32Z","title":"API-Bank: A Comprehensive Benchmark for Tool-Augmented LLMs","version":2},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-15T20:51:41.198767Z"},"links":{"cited_paper":"/paper/2107.03374","citing_paper":"/paper/2304.08244"},"observation_digest":"sha256:537f0086dfb2559b9bfed95823cb21f13e3ca82b03e83d30594c7a653a81aabe","observation_id":"ef3ea066-0f13-48cd-8186-9dc0bbfc0b41","resolution":{"observed_at":"2026-05-15T20:51:41.220201Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-01T08:08:23.404839+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-01T08:08:23.404839+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.11554","last_updated":"2024-01-15T23:52:21Z","snapshot_observed_at":"2026-08-07T10:54:19.055982Z","submitted_at":"2023-05-19T09:54:21Z","title":"ToolkenGPT: Augmenting Frozen Language Models with Massive Tools via Tool Embeddings","version":4},"cited_work":{"arxiv_id":"2305.11554","doi":"10.48550/arxiv.2305.11554","metadata_source":"arxiv_reference","pith_arxiv_id":"2305.11554","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv preprint arXiv:2305.11554","venue":"arXiv (Cornell University)","work_id":"569aa9d1-da86-4292-9955-f937133dafea","year":2023},"citing_paper":{"arxiv_id":"2304.08244","last_updated":"2023-10-25T06:54:12Z","snapshot_observed_at":"2026-08-02T00:07:12.855748Z","submitted_at":"2023-04-14T14:05:32Z","title":"API-Bank: A Comprehensive Benchmark for Tool-Augmented LLMs","version":2},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-05-15T20:51:41.198767Z"},"links":{"cited_paper":"/paper/2305.11554","citing_paper":"/paper/2304.08244"},"observation_digest":"sha256:bac6462b391808d35ffcac02c21a5fcbbd58e5712eb4e3beb64173355b702963","observation_id":"b5af33ed-2587-4617-8fdd-de8a36f0ca8d","resolution":{"observed_at":"2026-05-15T20:51:41.223381Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2208.03299","last_updated":"2022-11-16T16:38:18Z","snapshot_observed_at":"2026-08-05T00:57:17.859949Z","submitted_at":"2022-08-05T17:39:22Z","title":"Atlas: Few-shot Learning with Retrieval Augmented Language Models","version":3},"cited_work":{"arxiv_id":"2208.03299","doi":null,"metadata_source":"pith","pith_arxiv_id":"2208.03299","snapshot_observed_at":"2026-07-03T17:48:46.017911Z","title":"Atlas: Few-shot Learning with Retrieval Augmented Language Models","venue":"cs.CL","work_id":"3bffc484-91bd-41a2-9e31-922d0c311a12","year":2022},"citing_paper":{"arxiv_id":"2304.08244","last_updated":"2023-10-25T06:54:12Z","snapshot_observed_at":"2026-08-02T00:07:12.855748Z","submitted_at":"2023-04-14T14:05:32Z","title":"API-Bank: A Comprehensive Benchmark for Tool-Augmented LLMs","version":2},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-05-15T20:51:41.198767Z"},"links":{"cited_paper":"/paper/2208.03299","citing_paper":"/paper/2304.08244"},"observation_digest":"sha256:6393372416b63c8ac826d0cafadd8a32382033898e98b4f563a1d3ec3f23a73d","observation_id":"5156ba47-0c6d-492d-a45b-2e621718da54","resolution":{"observed_at":"2026-05-16T13:48:43.695710Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2303.16434","last_updated":"2023-03-29T03:30:38Z","snapshot_observed_at":"2026-07-06T15:09:21.780147Z","submitted_at":"2023-03-29T03:30:38Z","title":"TaskMatrix.AI: Completing Tasks by Connecting Foundation Models with Millions of APIs","version":1},"cited_work":{"arxiv_id":"2303.16434","doi":"10.48550/arxiv.2303.16434","metadata_source":"arxiv_reference","pith_arxiv_id":"2303.16434","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Taskmatrix.ai: Completing tasks by connecting foundation models with millions of apis","venue":"arXiv (Cornell University)","work_id":"90f98bad-b70c-481c-a56f-1197ee89e441","year":2023},"citing_paper":{"arxiv_id":"2304.08244","last_updated":"2023-10-25T06:54:12Z","snapshot_observed_at":"2026-08-02T00:07:12.855748Z","submitted_at":"2023-04-14T14:05:32Z","title":"API-Bank: A Comprehensive Benchmark for Tool-Augmented LLMs","version":2},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-05-15T20:51:41.198767Z"},"links":{"cited_paper":"/paper/2303.16434","citing_paper":"/paper/2304.08244"},"observation_digest":"sha256:9a8cdb2fb29af0031542b98e75ee5f3c275a3b77f7b7d472b8d553461c8e41bd","observation_id":"da151480-a189-4469-b097-b10dbc9d831b","resolution":{"observed_at":"2026-05-15T20:51:41.229535Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2302.07842","last_updated":"2023-02-15T18:25:52Z","snapshot_observed_at":"2026-07-30T02:11:00.198428Z","submitted_at":"2023-02-15T18:25:52Z","title":"Augmented Language Models: a Survey","version":1},"cited_work":{"arxiv_id":"2302.07842","doi":"10.48550/arxiv.2302.07842","metadata_source":"pith","pith_arxiv_id":"2302.07842","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Augmented Language Models: a Survey","venue":"cs.CL","work_id":"6426706e-f14a-4e4b-ade6-8414697a11d2","year":2023},"citing_paper":{"arxiv_id":"2304.08244","last_updated":"2023-10-25T06:54:12Z","snapshot_observed_at":"2026-08-02T00:07:12.855748Z","submitted_at":"2023-04-14T14:05:32Z","title":"API-Bank: A Comprehensive Benchmark for Tool-Augmented LLMs","version":2},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-05-15T20:51:41.198767Z"},"links":{"cited_paper":"/paper/2302.07842","citing_paper":"/paper/2304.08244"},"observation_digest":"sha256:c38ca4e537191fd486d81cb728963f1f2c0ed16a3271663506ad93ac608059e5","observation_id":"8299eb5f-16fb-4597-acc7-f89f102dd58a","resolution":{"observed_at":"2026-05-16T02:41:28.390682Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2112.09332","last_updated":"2022-06-01T19:08:11Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2021-12-17T05:43:43Z","title":"WebGPT: Browser-assisted question-answering with human feedback","version":3},"cited_work":{"arxiv_id":"2112.09332","doi":"10.48550/arxiv.2112.09332","metadata_source":"pith","pith_arxiv_id":"2112.09332","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"WebGPT: Browser-assisted question-answering with human feedback","venue":"cs.CL","work_id":"e25ef3e1-4848-4cb9-bf28-67a420591165","year":2021},"citing_paper":{"arxiv_id":"2304.08244","last_updated":"2023-10-25T06:54:12Z","snapshot_observed_at":"2026-08-02T00:07:12.855748Z","submitted_at":"2023-04-14T14:05:32Z","title":"API-Bank: A Comprehensive Benchmark for Tool-Augmented LLMs","version":2},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-05-15T20:51:41.198767Z"},"links":{"cited_paper":"/paper/2112.09332","citing_paper":"/paper/2304.08244"},"observation_digest":"sha256:c0a6f456742116f2527bbb2a921090abd1926d448970d5782acdc639787a7be5","observation_id":"250c8fff-585b-4489-a77f-d19bf8baa484","resolution":{"observed_at":"2026-05-15T20:51:41.235370Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-04T01:08:09.995583+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-04T01:08:09.995583+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2303.09014","last_updated":"2023-03-16T01:04:45Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-03-16T01:04:45Z","title":"ART: Automatic multi-step reasoning and tool-use for large language models","version":1},"cited_work":{"arxiv_id":"2303.09014","doi":"10.48550/arxiv.2303.09014","metadata_source":"pith","pith_arxiv_id":"2303.09014","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"ART: Automatic multi-step reasoning and tool-use for large language models","venue":"cs.CL","work_id":"0bffef46-48af-4851-abce-a3d6792d044b","year":2023},"citing_paper":{"arxiv_id":"2304.08244","last_updated":"2023-10-25T06:54:12Z","snapshot_observed_at":"2026-08-02T00:07:12.855748Z","submitted_at":"2023-04-14T14:05:32Z","title":"API-Bank: A Comprehensive Benchmark for Tool-Augmented LLMs","version":2},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-05-15T20:51:41.198767Z"},"links":{"cited_paper":"/paper/2303.09014","citing_paper":"/paper/2304.08244"},"observation_digest":"sha256:5046b3e64609ad6a3914ec1849a4cf141f8ac62e10a4457ec79a21566c19ed11","observation_id":"e7b489bd-d3ff-47e7-924e-5d54054b8650","resolution":{"observed_at":"2026-05-16T19:03:06.497846Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.15334","last_updated":"2023-05-24T16:48:11Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-24T16:48:11Z","title":"Gorilla: Large Language Model Connected with Massive APIs","version":1},"cited_work":{"arxiv_id":"2305.15334","doi":"10.48550/arxiv.2305.15334","metadata_source":"pith","pith_arxiv_id":"2305.15334","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Gorilla: Large Language Model Connected with Massive APIs","venue":"cs.CL","work_id":"126a464a-4a73-495f-b669-de1e44aa8f09","year":2023},"citing_paper":{"arxiv_id":"2304.08244","last_updated":"2023-10-25T06:54:12Z","snapshot_observed_at":"2026-08-02T00:07:12.855748Z","submitted_at":"2023-04-14T14:05:32Z","title":"API-Bank: A Comprehensive Benchmark for Tool-Augmented LLMs","version":2},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-05-15T20:51:41.198767Z"},"links":{"cited_paper":"/paper/2305.15334","citing_paper":"/paper/2304.08244"},"observation_digest":"sha256:79680d302824ddee9637f2e04a7bfff848fdd876738f008bc4601ccbf057b1b5","observation_id":"a5fb5a6a-5f0e-43ed-806d-e70ad41e22b0","resolution":{"observed_at":"2026-05-15T20:51:41.241440Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.14318","last_updated":"2024-06-21T16:51:22Z","snapshot_observed_at":"2026-08-05T00:55:07.906179Z","submitted_at":"2023-05-23T17:51:52Z","title":"CREATOR: Tool Creation for Disentangling Abstract and Concrete Reasoning of Large Language Models","version":3},"cited_work":{"arxiv_id":"2305.14318","doi":"10.48550/arxiv.2305.14318","metadata_source":"pith","pith_arxiv_id":"2305.14318","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Creator: Tool creation for disentangling abstract and concrete reasoning of large language models","venue":"cs.CL","work_id":"7b482a42-424e-4ba3-985c-54b3bff65f56","year":2023},"citing_paper":{"arxiv_id":"2304.08244","last_updated":"2023-10-25T06:54:12Z","snapshot_observed_at":"2026-08-02T00:07:12.855748Z","submitted_at":"2023-04-14T14:05:32Z","title":"API-Bank: A Comprehensive Benchmark for Tool-Augmented LLMs","version":2},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-05-15T20:51:41.198767Z"},"links":{"cited_paper":"/paper/2305.14318","citing_paper":"/paper/2304.08244"},"observation_digest":"sha256:2256404cb3e549bdfdf2935af2a33546a63df46ca245c7d1d858923259b87577","observation_id":"978e2a16-ee6e-4f7b-b841-2adbd758b011","resolution":{"observed_at":"2026-05-15T20:51:41.244375Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.13068","last_updated":"2024-03-14T08:15:27Z","snapshot_observed_at":"2026-07-06T15:30:42.173342Z","submitted_at":"2023-05-22T14:37:05Z","title":"Making Language Models Better Tool Learners with Execution Feedback","version":3},"cited_work":{"arxiv_id":"2305.13068","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2305.13068","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Making language models better tool learners with execution feedback","venue":null,"work_id":"1641aa28-2905-46ff-b378-7424f415e0ca","year":2023},"citing_paper":{"arxiv_id":"2304.08244","last_updated":"2023-10-25T06:54:12Z","snapshot_observed_at":"2026-08-02T00:07:12.855748Z","submitted_at":"2023-04-14T14:05:32Z","title":"API-Bank: A Comprehensive Benchmark for Tool-Augmented LLMs","version":2},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-05-15T20:51:41.198767Z"},"links":{"cited_paper":"/paper/2305.13068","citing_paper":"/paper/2304.08244"},"observation_digest":"sha256:cdf110cc211b9e3d85e2e8b85affec51e765bfcd0fbc2264a3764978f4589ee7","observation_id":"70825f6d-7bd5-4910-acb7-0d9e37811786","resolution":{"observed_at":"2026-05-15T20:51:41.247322Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2302.04761","last_updated":"2023-02-09T16:49:57Z","snapshot_observed_at":"2026-07-06T14:50:07.491434Z","submitted_at":"2023-02-09T16:49:57Z","title":"Toolformer: Language Models Can Teach Themselves to Use Tools","version":1},"cited_work":{"arxiv_id":"2302.04761","doi":"10.48550/arxiv.2302.04761","metadata_source":"pith","pith_arxiv_id":"2302.04761","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Toolformer: Language Models Can Teach Themselves to Use Tools","venue":"cs.CL","work_id":"9bce40c8-cfd7-4983-80e0-c3bd4402322a","year":2023},"citing_paper":{"arxiv_id":"2304.08244","last_updated":"2023-10-25T06:54:12Z","snapshot_observed_at":"2026-08-02T00:07:12.855748Z","submitted_at":"2023-04-14T14:05:32Z","title":"API-Bank: A Comprehensive Benchmark for Tool-Augmented LLMs","version":2},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-05-15T20:51:41.198767Z"},"links":{"cited_paper":"/paper/2302.04761","citing_paper":"/paper/2304.08244"},"observation_digest":"sha256:d1fa5f5ea14cdec4a2d1bd46dd579272bf41d728ea4e424ff90e534eec50ac0b","observation_id":"156bed37-3106-45fe-a0c1-bbf1decb5782","resolution":{"observed_at":"2026-05-15T20:51:41.250457Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-04T01:08:11.107254+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-04T01:08:11.107254+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.17492","last_updated":"2024-02-27T18:42:42Z","snapshot_observed_at":"2026-08-02T19:39:02.664535Z","submitted_at":"2023-06-30T09:07:37Z","title":"Preference Ranking Optimization for Human Alignment","version":2},"cited_work":{"arxiv_id":"2306.17492","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2306.17492","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Zayne Sprague, Xi Ye, Kaj Bostrom, Swarat Chaudhuri, and Greg Durrett","venue":null,"work_id":"e5243260-1070-4183-9c7f-287ebd73c702","year":2023},"citing_paper":{"arxiv_id":"2304.08244","last_updated":"2023-10-25T06:54:12Z","snapshot_observed_at":"2026-08-02T00:07:12.855748Z","submitted_at":"2023-04-14T14:05:32Z","title":"API-Bank: A Comprehensive Benchmark for Tool-Augmented LLMs","version":2},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-05-15T20:51:41.198767Z"},"links":{"cited_paper":"/paper/2306.17492","citing_paper":"/paper/2304.08244"},"observation_digest":"sha256:e36b816eeccce9b38af8d9f481a1501ca8197ba458fcecb32e780584b0463ceb","observation_id":"40d25fbf-7ebd-45be-9e3e-a9900ab27456","resolution":{"observed_at":"2026-05-15T20:51:41.253387Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.05301","last_updated":"2023-09-07T12:20:45Z","snapshot_observed_at":"2026-07-06T15:40:20.267344Z","submitted_at":"2023-06-08T15:46:32Z","title":"ToolAlpaca: Generalized Tool Learning for Language Models with 3000 Simulated Cases","version":2},"cited_work":{"arxiv_id":"2306.05301","doi":"10.48550/arxiv.2306.05301","metadata_source":"pith","pith_arxiv_id":"2306.05301","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"ToolAlpaca: Generalized Tool Learning for Language Models with 3000 Simulated Cases","venue":"cs.CL","work_id":"e900f660-9178-4fda-ad54-a788e23aa0d8","year":2023},"citing_paper":{"arxiv_id":"2304.08244","last_updated":"2023-10-25T06:54:12Z","snapshot_observed_at":"2026-08-02T00:07:12.855748Z","submitted_at":"2023-04-14T14:05:32Z","title":"API-Bank: A Comprehensive Benchmark for Tool-Augmented LLMs","version":2},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-05-15T20:51:41.198767Z"},"links":{"cited_paper":"/paper/2306.05301","citing_paper":"/paper/2304.08244"},"observation_digest":"sha256:f530c3c46c4fda8369aa91e2e1f43c6aad3cd81add708d769f48c5fbbbf0329d","observation_id":"6a311f75-1698-456f-8832-8f53030f1dbd","resolution":{"observed_at":"2026-05-15T23:03:48.638764Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2302.13971","last_updated":"2023-02-27T17:11:15Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-02-27T17:11:15Z","title":"LLaMA: Open and Efficient Foundation Language Models","version":1},"cited_work":{"arxiv_id":"2302.13971","doi":"10.48550/arxiv.2302.13971","metadata_source":"pith","pith_arxiv_id":"2302.13971","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"LLaMA: Open and Efficient Foundation Language Models","venue":"cs.CL","work_id":"c018fc23-6f3f-4035-9d02-28a2173b2b9d","year":2023},"citing_paper":{"arxiv_id":"2304.08244","last_updated":"2023-10-25T06:54:12Z","snapshot_observed_at":"2026-08-02T00:07:12.855748Z","submitted_at":"2023-04-14T14:05:32Z","title":"API-Bank: A Comprehensive Benchmark for Tool-Augmented LLMs","version":2},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-05-15T20:51:41.198767Z"},"links":{"cited_paper":"/paper/2302.13971","citing_paper":"/paper/2304.08244"},"observation_digest":"sha256:6840ed99088799f80b8c8596499c594f975dc0091143bdcc79dd5eb7cd0434c3","observation_id":"ba776283-7bd2-4bbc-bf83-38c71dffbe25","resolution":{"observed_at":"2026-05-15T20:51:41.259217Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-01T11:08:05.851253+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-01T11:08:05.851253+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2212.10560","last_updated":"2023-05-25T23:50:07Z","snapshot_observed_at":"2026-07-06T14:33:11.945106Z","submitted_at":"2022-12-20T18:59:19Z","title":"Self-Instruct: Aligning Language Models with Self-Generated Instructions","version":2},"cited_work":{"arxiv_id":"2212.10560","doi":"10.1145/3209978.3210080","metadata_source":"pith","pith_arxiv_id":"2212.10560","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Self-Instruct: Aligning Language Models with Self-Generated Instructions","venue":"cs.CL","work_id":"d0018767-775d-406e-861d-539ed681ff73","year":2022},"citing_paper":{"arxiv_id":"2304.08244","last_updated":"2023-10-25T06:54:12Z","snapshot_observed_at":"2026-08-02T00:07:12.855748Z","submitted_at":"2023-04-14T14:05:32Z","title":"API-Bank: A Comprehensive Benchmark for Tool-Augmented LLMs","version":2},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-05-15T20:51:41.198767Z"},"links":{"cited_paper":"/paper/2212.10560","citing_paper":"/paper/2304.08244"},"observation_digest":"sha256:7a37c02e704f5f5cb9a2e1153e1c56e5637958759c7bb6260de01c7030249a4b","observation_id":"bfbdf7ce-c539-4cbb-bd42-0af8fbe72369","resolution":{"observed_at":"2026-05-15T20:51:41.261979Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-05-25T23:23:20.678+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-25T23:23:20.678+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2210.03629","last_updated":"2023-03-10T01:00:17Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2022-10-06T01:00:32Z","title":"ReAct: Synergizing Reasoning and Acting in Language Models","version":3},"cited_work":{"arxiv_id":"2210.03629","doi":"10.48550/arxiv.2210.03629","metadata_source":"pith","pith_arxiv_id":"2210.03629","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"ReAct: Synergizing Reasoning and Acting in Language Models","venue":"cs.CL","work_id":"407a2351-25f1-497d-b611-f77d0292a8e6","year":2022},"citing_paper":{"arxiv_id":"2304.08244","last_updated":"2023-10-25T06:54:12Z","snapshot_observed_at":"2026-08-02T00:07:12.855748Z","submitted_at":"2023-04-14T14:05:32Z","title":"API-Bank: A Comprehensive Benchmark for Tool-Augmented LLMs","version":2},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-05-15T20:51:41.198767Z"},"links":{"cited_paper":"/paper/2210.03629","citing_paper":"/paper/2304.08244"},"observation_digest":"sha256:5442fada622991d440669f76f0e7879a74b74f46e326ddd0037c67acba73df19","observation_id":"f235feca-3722-451c-9108-83b41a289ca4","resolution":{"observed_at":"2026-05-15T20:51:41.264870Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-07-12T03:19:36.897515+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-12T03:19:36.897515+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2204.00598","last_updated":"2022-05-27T17:52:50Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2022-04-01T17:43:13Z","title":"Socratic Models: Composing Zero-Shot Multimodal Reasoning with Language","version":2},"cited_work":{"arxiv_id":"2204.00598","doi":"10.48550/arxiv.2204.00598","metadata_source":"pith","pith_arxiv_id":"2204.00598","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Socratic Models: Composing Zero-Shot Multimodal Reasoning with Language","venue":"cs.CV","work_id":"6c2a6dd5-9b0f-4d86-a291-b605ebdfde6c","year":2022},"citing_paper":{"arxiv_id":"2304.08244","last_updated":"2023-10-25T06:54:12Z","snapshot_observed_at":"2026-08-02T00:07:12.855748Z","submitted_at":"2023-04-14T14:05:32Z","title":"API-Bank: A Comprehensive Benchmark for Tool-Augmented LLMs","version":2},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-05-15T20:51:41.198767Z"},"links":{"cited_paper":"/paper/2204.00598","citing_paper":"/paper/2304.08244"},"observation_digest":"sha256:931527d314d710708ff5fc354a22cd21ca5281f9b708de26c4072b6b48175858","observation_id":"a1cb8f7d-c6b6-4807-910f-338c384ceca0","resolution":{"observed_at":"2026-05-16T09:50:01.041344Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.05696","last_updated":"2024-02-29T03:04:22Z","snapshot_observed_at":"2026-07-06T16:04:58.235073Z","submitted_at":"2023-08-10T16:58:51Z","title":"A Preliminary Study of the Intrinsic Relationship between Complexity and Alignment","version":2},"cited_work":{"arxiv_id":"2308.05696","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2308.05696","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"arXiv preprint arXiv:2308.05696","venue":null,"work_id":"58821c4e-73bd-456c-acbd-7072002f30ac","year":2023},"citing_paper":{"arxiv_id":"2304.08244","last_updated":"2023-10-25T06:54:12Z","snapshot_observed_at":"2026-08-02T00:07:12.855748Z","submitted_at":"2023-04-14T14:05:32Z","title":"API-Bank: A Comprehensive Benchmark for Tool-Augmented LLMs","version":2},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-05-15T20:51:41.198767Z"},"links":{"cited_paper":"/paper/2308.05696","citing_paper":"/paper/2304.08244"},"observation_digest":"sha256:d856dc8efdfc24751887274bc4a25660cb4ea5dc6a69f293b390ef86d8a3bbd3","observation_id":"ce24e540-d6f4-49c7-8ed1-be0308682643","resolution":{"observed_at":"2026-05-15T20:51:41.270956Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.13304","last_updated":"2023-06-23T05:43:28Z","snapshot_observed_at":"2026-07-06T15:45:51.145364Z","submitted_at":"2023-06-23T05:43:28Z","title":"ToolQA: A Dataset for LLM Question Answering with External Tools","version":1},"cited_work":{"arxiv_id":"2306.13304","doi":"10.48550/arxiv.2306.13304","metadata_source":"arxiv_reference","pith_arxiv_id":"2306.13304","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv preprint arXiv:2306.13304","venue":"arXiv (Cornell University)","work_id":"6ed7e375-413b-460b-8d3c-506cc289cfbb","year":2023},"citing_paper":{"arxiv_id":"2304.08244","last_updated":"2023-10-25T06:54:12Z","snapshot_observed_at":"2026-08-02T00:07:12.855748Z","submitted_at":"2023-04-14T14:05:32Z","title":"API-Bank: A Comprehensive Benchmark for Tool-Augmented LLMs","version":2},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-05-15T20:51:41.198767Z"},"links":{"cited_paper":"/paper/2306.13304","citing_paper":"/paper/2304.08244"},"observation_digest":"sha256:836c245198bd17d58cbc1e76950c7f9508d40a4e1c421d046accecbcc7238a2f","observation_id":"44c9d07d-b1b4-4e14-9e1f-d4cfcb900f9b","resolution":{"observed_at":"2026-05-15T20:51:41.210477Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"name\": \"ToolSearcher","venue":null,"work_id":"94f606d9-6650-4afe-a321-bc98cd88183c","year":2023},"citing_paper":{"arxiv_id":"2304.08244","last_updated":"2023-10-25T06:54:12Z","snapshot_observed_at":"2026-08-02T00:07:12.855748Z","submitted_at":"2023-04-14T14:05:32Z","title":"API-Bank: A Comprehensive Benchmark for Tool-Augmented LLMs","version":2},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-05-15T20:51:41.198767Z"},"links":{"citing_paper":"/paper/2304.08244"},"observation_digest":"sha256:46a8d39f51a696ba191cf8943a85dd1b8e5657eb5bd699de688e64f19d412d69","observation_id":"bf9fc090-dc9b-459a-84de-acffc165fabd","resolution":{"observed_at":"2026-05-15T20:51:41.273003Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2304.08244","last_updated":"2023-10-25T06:54:12Z","latest_version":2,"primary_category":"cs.CL","snapshot_observed_at":"2026-08-02T00:07:12.855748Z","submitted_at":"2023-04-14T14:05:32Z","title":"API-Bank: A Comprehensive Benchmark for Tool-Augmented LLMs"},"reference_resolution":{"displayed":23,"state_counts":{"malformed_identifier":0,"metadata_mismatch":17,"parse_uncertain":0,"unresolved":0,"verified_exact":4,"verified_fuzzy":2},"total_outbound_references":23},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"thesis":"As of 7 August 2026, this Paper Citation Record lists 23 of 23 outbound references and 81 inbound Pith citation observations for arXiv:2304.08244."}