{"as_of":"2026-08-23T21:17:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:e8ea6a76bcc4a05deea8c76acbb3afa96f8440de02ef0e3d2aefcd88aaaf1083","coverage":[{"denominator":58,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":58,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-15T20:21:22.416314Z","state":"measured"},{"denominator":60,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":60,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-23T06:30:58.430688+00:00","state":"measured"},{"denominator":2,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":2,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-05T17:56:47.089599Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"pith","source_observed_at":"2026-08-05T17:56:51.738856Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2505.13328","last_updated":"2025-05-19T16:36:13Z","snapshot_observed_at":"2026-08-23T02:31:24.768215Z","submitted_at":"2025-05-19T16:36:13Z","title":"Rethinking Stateful Tool Use in Multi-Turn Dialogues: Benchmarks and Challenges","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2505.13328","snapshot_observed_at":"2026-08-03T18:03:38.427855Z","title":"Z., and Wong, K.-F","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2512.07287","last_updated":"2026-06-28T11:58:42Z","snapshot_observed_at":"2026-08-16T17:55:23.212702Z","submitted_at":"2025-12-08T08:27:24Z","title":"Experience-Evolving Multi-Turn Tool-Use Agent with Hybrid Episodic-Procedural Memory","version":3},"reference_index":2023,"source":"pdf_text","source_observed_at":"2026-08-03T18:03:38.427855Z"},"links":{"cited_paper":"/paper/2505.13328","citing_paper":"/paper/2512.07287"},"observation_digest":"sha256:2ab7cc409dc077eb06a7a8014fa58ce3d4c322a67b5f641db4e1e25a25337ac2","observation_id":"24699e33-12a6-4b9e-89d9-cf52684af9ce","resolution":{"observed_at":"2026-08-03T18:03:38.427855Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2505.13328","last_updated":"2025-05-19T16:36:13Z","snapshot_observed_at":"2026-08-23T02:31:24.768215Z","submitted_at":"2025-05-19T16:36:13Z","title":"Rethinking Stateful Tool Use in Multi-Turn Dialogues: Benchmarks and Challenges","version":1},"cited_work":{"arxiv_id":"2505.13328","doi":null,"metadata_source":"pith","pith_arxiv_id":"2505.13328","snapshot_observed_at":"2026-08-05T17:56:51.738856Z","title":"Rethinking Stateful Tool Use in Multi-Turn Dialogues: Benchmarks and Challenges","venue":"cs.CL","work_id":"2db31445-0af5-4dae-beb4-64cb71ba9d6f","year":2025},"citing_paper":{"arxiv_id":"2608.03499","last_updated":"2026-08-04T11:42:26Z","snapshot_observed_at":"2026-08-14T05:40:05.736283Z","submitted_at":"2026-08-04T11:42:26Z","title":"WeClawArena: An Auditable Sandbox and Benchmark for Cross-User Agents Collaboration and Security in Human-Centered Agent Networks","version":1},"reference_index":14,"source":"arxiv_source","source_observed_at":"2026-08-05T17:56:47.089599Z"},"links":{"cited_paper":"/paper/2505.13328","citing_paper":"/paper/2608.03499"},"observation_digest":"sha256:040688c3767977519e055df60d18d4ce4e893c18891354353465892037816897","observation_id":"7b16a860-d196-4d13-ae6c-60fa27126767","resolution":{"observed_at":"2026-08-05T17:56:51.794005Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2505.13328/citation-record","integrity":"/paper/2505.13328/integrity","json":"/paper/2505.13328/citation-record.json","paper":"/paper/2505.13328"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T20:21:21.964283Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.13328","last_updated":"2025-05-19T16:36:13Z","snapshot_observed_at":"2026-08-23T02:31:24.768215Z","submitted_at":"2025-05-19T16:36:13Z","title":"Rethinking Stateful Tool Use in Multi-Turn Dialogues: Benchmarks and Challenges","version":1},"reference_index":1,"source":"arxiv_source","source_observed_at":"2026-08-15T20:21:21.964283Z"},"links":{"citing_paper":"/paper/2505.13328"},"observation_digest":"sha256:46bc43214ae7aed475ad91cc8ae601a3af8caf430084fca37a7ae41aa5f5895f","observation_id":"bfd89ea3-3418-41db-906e-68600786822b","resolution":{"observed_at":"2026-08-15T20:21:21.964283Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2309.16609","last_updated":"2023-09-28T17:07:49Z","snapshot_observed_at":"2026-08-20T15:31:01.041088Z","submitted_at":"2023-09-28T17:07:49Z","title":"Qwen Technical Report","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.16609","snapshot_observed_at":"2026-08-15T20:21:21.970313Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.13328","last_updated":"2025-05-19T16:36:13Z","snapshot_observed_at":"2026-08-23T02:31:24.768215Z","submitted_at":"2025-05-19T16:36:13Z","title":"Rethinking Stateful Tool Use in Multi-Turn Dialogues: Benchmarks and Challenges","version":1},"reference_index":2,"source":"arxiv_source","source_observed_at":"2026-08-15T20:21:21.970313Z"},"links":{"cited_paper":"/paper/2309.16609","citing_paper":"/paper/2505.13328"},"observation_digest":"sha256:7747bdcce83b60e6df69e361d885505774cd09e3696f19877219d6fc22654e80","observation_id":"976a5420-ddfe-4684-88cc-c55c45e6e571","resolution":{"observed_at":"2026-08-15T20:21:21.970313Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T20:21:23.460543Z","title":null,"venue":null,"work_id":"8de02a88-1685-4866-ba53-5a0dcdb11783","year":1991},"citing_paper":{"arxiv_id":"2505.13328","last_updated":"2025-05-19T16:36:13Z","snapshot_observed_at":"2026-08-23T02:31:24.768215Z","submitted_at":"2025-05-19T16:36:13Z","title":"Rethinking Stateful Tool Use in Multi-Turn Dialogues: Benchmarks and Challenges","version":1},"reference_index":3,"source":"arxiv_source","source_observed_at":"2026-08-15T20:21:21.975760Z"},"links":{"citing_paper":"/paper/2505.13328"},"observation_digest":"sha256:12af72d6ce25138aed25e3e6a75b0f116847c5ef478ef536451b4a9ec96d1f22","observation_id":"f001b702-622b-4863-bf3a-035bf3b52f45","resolution":{"observed_at":"2026-08-15T20:21:23.467081Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T20:21:21.980696Z","title":null,"venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2505.13328","last_updated":"2025-05-19T16:36:13Z","snapshot_observed_at":"2026-08-23T02:31:24.768215Z","submitted_at":"2025-05-19T16:36:13Z","title":"Rethinking Stateful Tool Use in Multi-Turn Dialogues: Benchmarks and Challenges","version":1},"reference_index":4,"source":"arxiv_source","source_observed_at":"2026-08-15T20:21:21.980696Z"},"links":{"citing_paper":"/paper/2505.13328"},"observation_digest":"sha256:0aaa99cbd0e52d73739b1c64f56c8ff3ad7279f23e09cdfd2a548e3c91549401","observation_id":"0434f184-a0fe-44db-a97c-f23f13e81ed4","resolution":{"observed_at":"2026-08-15T20:21:21.980696Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"1811.01241","last_updated":"2019-02-21T22:23:34Z","snapshot_observed_at":"2026-08-14T18:04:44.707582Z","submitted_at":"2018-11-03T16:11:29Z","title":"Wizard of Wikipedia: Knowledge-Powered Conversational agents","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"1811.01241","snapshot_observed_at":"2026-08-15T20:21:21.985873Z","title":null,"venue":null,"work_id":null,"year":2018},"citing_paper":{"arxiv_id":"2505.13328","last_updated":"2025-05-19T16:36:13Z","snapshot_observed_at":"2026-08-23T02:31:24.768215Z","submitted_at":"2025-05-19T16:36:13Z","title":"Rethinking Stateful Tool Use in Multi-Turn Dialogues: Benchmarks and Challenges","version":1},"reference_index":5,"source":"arxiv_source","source_observed_at":"2026-08-15T20:21:21.985873Z"},"links":{"cited_paper":"/paper/1811.01241","citing_paper":"/paper/2505.13328"},"observation_digest":"sha256:e3f3468324708d19c92eafa643167db6768a3141c7c07450878e9683923465b8","observation_id":"97cf81e5-b89d-4e27-ac98-cae7a2a73a76","resolution":{"observed_at":"2026-08-15T20:21:21.985873Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T20:21:21.991269Z","title":null,"venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2505.13328","last_updated":"2025-05-19T16:36:13Z","snapshot_observed_at":"2026-08-23T02:31:24.768215Z","submitted_at":"2025-05-19T16:36:13Z","title":"Rethinking Stateful Tool Use in Multi-Turn Dialogues: Benchmarks and Challenges","version":1},"reference_index":6,"source":"arxiv_source","source_observed_at":"2026-08-15T20:21:21.991269Z"},"links":{"citing_paper":"/paper/2505.13328"},"observation_digest":"sha256:896a82eb17a883fad1af846a0eb9d368b328b5a8282b62ce71227d1f2845e3e7","observation_id":"9c122317-852f-41a7-af9d-acd353c862b5","resolution":{"observed_at":"2026-08-15T20:21:21.991269Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2312.03815","last_updated":"2023-12-09T18:10:39Z","snapshot_observed_at":"2026-08-21T16:02:46.082950Z","submitted_at":"2023-12-06T18:50:26Z","title":"LLM as OS, Agents as Apps: Envisioning AIOS, Agents and the AIOS-Agent Ecosystem","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2312.03815","snapshot_observed_at":"2026-08-15T20:21:21.996407Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.13328","last_updated":"2025-05-19T16:36:13Z","snapshot_observed_at":"2026-08-23T02:31:24.768215Z","submitted_at":"2025-05-19T16:36:13Z","title":"Rethinking Stateful Tool Use in Multi-Turn Dialogues: Benchmarks and Challenges","version":1},"reference_index":7,"source":"arxiv_source","source_observed_at":"2026-08-15T20:21:21.996407Z"},"links":{"cited_paper":"/paper/2312.03815","citing_paper":"/paper/2505.13328"},"observation_digest":"sha256:028c3c729e75f7188656f604d436a007694bf9ab9743086aa9326a54625825a1","observation_id":"e8a983d1-90a0-4947-9e68-45745d6dbc24","resolution":{"observed_at":"2026-08-15T20:21:21.996407Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T20:21:22.001578Z","title":null,"venue":null,"work_id":null,"year":1991},"citing_paper":{"arxiv_id":"2505.13328","last_updated":"2025-05-19T16:36:13Z","snapshot_observed_at":"2026-08-23T02:31:24.768215Z","submitted_at":"2025-05-19T16:36:13Z","title":"Rethinking Stateful Tool Use in Multi-Turn Dialogues: Benchmarks and Challenges","version":1},"reference_index":8,"source":"arxiv_source","source_observed_at":"2026-08-15T20:21:22.001578Z"},"links":{"citing_paper":"/paper/2505.13328"},"observation_digest":"sha256:92ea1e67522a90fbad4962411906c4ae52900e52660d588bef112066e68342f5","observation_id":"233a01b5-6ff7-40ec-b1ba-b3cf6dad6982","resolution":{"observed_at":"2026-08-15T20:21:22.001578Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T20:21:23.414057Z","title":null,"venue":null,"work_id":"5f6958ca-3463-4899-83e1-9e3f3c02d492","year":2018},"citing_paper":{"arxiv_id":"2505.13328","last_updated":"2025-05-19T16:36:13Z","snapshot_observed_at":"2026-08-23T02:31:24.768215Z","submitted_at":"2025-05-19T16:36:13Z","title":"Rethinking Stateful Tool Use in Multi-Turn Dialogues: Benchmarks and Challenges","version":1},"reference_index":9,"source":"arxiv_source","source_observed_at":"2026-08-15T20:21:22.006135Z"},"links":{"citing_paper":"/paper/2505.13328"},"observation_digest":"sha256:9c621fb8e3103934a86d7fa9871265460e553355bcf2a50462a817f8c8f23f11","observation_id":"8152d58d-6f9f-46a9-addc-ef5a90babe9e","resolution":{"observed_at":"2026-08-15T20:21:23.419766Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.17167","last_updated":"2024-06-03T11:28:29Z","snapshot_observed_at":"2026-08-16T14:23:10.558509Z","submitted_at":"2024-01-30T16:52:56Z","title":"Planning, Creation, Usage: Benchmarking LLMs for Comprehensive Tool Utilization in Real-World Complex Scenarios","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.17167","snapshot_observed_at":"2026-08-15T20:21:22.010719Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.13328","last_updated":"2025-05-19T16:36:13Z","snapshot_observed_at":"2026-08-23T02:31:24.768215Z","submitted_at":"2025-05-19T16:36:13Z","title":"Rethinking Stateful Tool Use in Multi-Turn Dialogues: Benchmarks and Challenges","version":1},"reference_index":10,"source":"arxiv_source","source_observed_at":"2026-08-15T20:21:22.010719Z"},"links":{"cited_paper":"/paper/2401.17167","citing_paper":"/paper/2505.13328"},"observation_digest":"sha256:e4627a15ddd8abe85a46460388c2a613cf92905d740e523b6ee9c77430f13a63","observation_id":"93261e84-8f4f-4cb4-bee5-3f8229532169","resolution":{"observed_at":"2026-08-15T20:21:22.010719Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2201.07207","last_updated":"2022-03-08T06:47:17Z","snapshot_observed_at":"2026-08-20T01:44:55.318386Z","submitted_at":"2022-01-18T18:59:45Z","title":"Language Models as Zero-Shot Planners: Extracting Actionable Knowledge for Embodied Agents","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2201.07207","snapshot_observed_at":"2026-08-15T20:21:22.016250Z","title":null,"venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2505.13328","last_updated":"2025-05-19T16:36:13Z","snapshot_observed_at":"2026-08-23T02:31:24.768215Z","submitted_at":"2025-05-19T16:36:13Z","title":"Rethinking Stateful Tool Use in Multi-Turn Dialogues: Benchmarks and Challenges","version":1},"reference_index":11,"source":"arxiv_source","source_observed_at":"2026-08-15T20:21:22.016250Z"},"links":{"cited_paper":"/paper/2201.07207","citing_paper":"/paper/2505.13328"},"observation_digest":"sha256:54be3c31ccb2155fb92662d13889756cdd88a9e9f88b5472a96a4adad1873918","observation_id":"45bb2c7e-e636-4759-b1d1-b7414dccb115","resolution":{"observed_at":"2026-08-15T20:21:22.016250Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T20:21:22.020973Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.13328","last_updated":"2025-05-19T16:36:13Z","snapshot_observed_at":"2026-08-23T02:31:24.768215Z","submitted_at":"2025-05-19T16:36:13Z","title":"Rethinking Stateful Tool Use in Multi-Turn Dialogues: Benchmarks and Challenges","version":1},"reference_index":12,"source":"arxiv_source","source_observed_at":"2026-08-15T20:21:22.020973Z"},"links":{"citing_paper":"/paper/2505.13328"},"observation_digest":"sha256:5f3a30958125e5b1b6c69a861d1a014efef780d28f4b56296265e05fff911d8a","observation_id":"ebd0970c-d357-4967-a142-175dc3457a0c","resolution":{"observed_at":"2026-08-15T20:21:22.020973Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.06825","last_updated":"2023-10-10T17:54:58Z","snapshot_observed_at":"2026-08-17T20:30:34.016254Z","submitted_at":"2023-10-10T17:54:58Z","title":"Mistral 7B","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.06825","snapshot_observed_at":"2026-08-15T20:21:22.025948Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.13328","last_updated":"2025-05-19T16:36:13Z","snapshot_observed_at":"2026-08-23T02:31:24.768215Z","submitted_at":"2025-05-19T16:36:13Z","title":"Rethinking Stateful Tool Use in Multi-Turn Dialogues: Benchmarks and Challenges","version":1},"reference_index":13,"source":"arxiv_source","source_observed_at":"2026-08-15T20:21:22.025948Z"},"links":{"cited_paper":"/paper/2310.06825","citing_paper":"/paper/2505.13328"},"observation_digest":"sha256:538580fa35364908934cae30bd31a6ffc83c114d8ddd697f13914661e3c6ecc7","observation_id":"5d27af6e-1617-4e5d-b691-ed7dd0ab89b6","resolution":{"observed_at":"2026-08-15T20:21:22.025948Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T20:21:22.030598Z","title":null,"venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2505.13328","last_updated":"2025-05-19T16:36:13Z","snapshot_observed_at":"2026-08-23T02:31:24.768215Z","submitted_at":"2025-05-19T16:36:13Z","title":"Rethinking Stateful Tool Use in Multi-Turn Dialogues: Benchmarks and Challenges","version":1},"reference_index":14,"source":"arxiv_source","source_observed_at":"2026-08-15T20:21:22.030598Z"},"links":{"citing_paper":"/paper/2505.13328"},"observation_digest":"sha256:60a3add78e12f2cf8030bb9eeb0ffe017d29c1fd17c8af5b4a123044919a7dfe","observation_id":"a89fcbfc-176f-4ec6-81de-d3c14286f919","resolution":{"observed_at":"2026-08-15T20:21:22.030598Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T20:21:23.397773Z","title":null,"venue":null,"work_id":"c047447d-977a-4648-a048-da9b700105c4","year":2024},"citing_paper":{"arxiv_id":"2505.13328","last_updated":"2025-05-19T16:36:13Z","snapshot_observed_at":"2026-08-23T02:31:24.768215Z","submitted_at":"2025-05-19T16:36:13Z","title":"Rethinking Stateful Tool Use in Multi-Turn Dialogues: Benchmarks and Challenges","version":1},"reference_index":15,"source":"arxiv_source","source_observed_at":"2026-08-15T20:21:22.035473Z"},"links":{"citing_paper":"/paper/2505.13328"},"observation_digest":"sha256:66a0832271596ed3294a08ea6c232877a0cd9bff2b2cda131f81e9726924d998","observation_id":"c69dc4e2-cbb7-42e9-8617-da2224961602","resolution":{"observed_at":"2026-08-15T20:21:23.402571Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2005.11401","last_updated":"2021-04-12T15:42:18Z","snapshot_observed_at":"2026-08-07T05:44:30.677502Z","submitted_at":"2020-05-22T21:34:34Z","title":"Retrieval-Augmented Generation for Knowledge-Intensive NLP Tasks","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2005.11401","snapshot_observed_at":"2026-08-15T20:21:22.040140Z","title":null,"venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2505.13328","last_updated":"2025-05-19T16:36:13Z","snapshot_observed_at":"2026-08-23T02:31:24.768215Z","submitted_at":"2025-05-19T16:36:13Z","title":"Rethinking Stateful Tool Use in Multi-Turn Dialogues: Benchmarks and Challenges","version":1},"reference_index":16,"source":"arxiv_source","source_observed_at":"2026-08-15T20:21:22.040140Z"},"links":{"cited_paper":"/paper/2005.11401","citing_paper":"/paper/2505.13328"},"observation_digest":"sha256:170780eb52bfe2799f6b62d3a76daffdc2f22ed9e45cf9f10c629bed16760148","observation_id":"1c4b58d0-74b6-49d1-aa6f-29a04d5a8ecd","resolution":{"observed_at":"2026-08-15T20:21:22.040140Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T20:21:22.048321Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.13328","last_updated":"2025-05-19T16:36:13Z","snapshot_observed_at":"2026-08-23T02:31:24.768215Z","submitted_at":"2025-05-19T16:36:13Z","title":"Rethinking Stateful Tool Use in Multi-Turn Dialogues: Benchmarks and Challenges","version":1},"reference_index":17,"source":"arxiv_source","source_observed_at":"2026-08-15T20:21:22.048321Z"},"links":{"citing_paper":"/paper/2505.13328"},"observation_digest":"sha256:6351246ded0062738a1744f1db1940167f88d68ac5550e45710840b45042a1ee","observation_id":"3363668e-9a48-4983-87b3-8f058e75f83e","resolution":{"observed_at":"2026-08-15T20:21:22.048321Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2209.07753","last_updated":"2023-05-25T03:50:11Z","snapshot_observed_at":"2026-08-21T12:10:42.882805Z","submitted_at":"2022-09-16T07:17:23Z","title":"Code as Policies: Language Model Programs for Embodied Control","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2209.07753","snapshot_observed_at":"2026-08-15T20:21:22.094572Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.13328","last_updated":"2025-05-19T16:36:13Z","snapshot_observed_at":"2026-08-23T02:31:24.768215Z","submitted_at":"2025-05-19T16:36:13Z","title":"Rethinking Stateful Tool Use in Multi-Turn Dialogues: Benchmarks and Challenges","version":1},"reference_index":18,"source":"arxiv_source","source_observed_at":"2026-08-15T20:21:22.094572Z"},"links":{"cited_paper":"/paper/2209.07753","citing_paper":"/paper/2505.13328"},"observation_digest":"sha256:ef239e019d9c9e2b2428434ea8d7c9f5faeb4330cdfd6d54e4caf62228ab3606","observation_id":"8c291d8c-cbf8-440a-bd2c-07e2fa8c7a0f","resolution":{"observed_at":"2026-08-15T20:21:22.094572Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2308.03688","last_updated":"2025-10-04T03:54:18Z","snapshot_observed_at":"2026-08-20T10:21:12.032735Z","submitted_at":"2023-08-07T16:08:11Z","title":"AgentBench: Evaluating LLMs as Agents","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2308.03688","snapshot_observed_at":"2026-08-15T20:21:22.143383Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.13328","last_updated":"2025-05-19T16:36:13Z","snapshot_observed_at":"2026-08-23T02:31:24.768215Z","submitted_at":"2025-05-19T16:36:13Z","title":"Rethinking Stateful Tool Use in Multi-Turn Dialogues: Benchmarks and Challenges","version":1},"reference_index":19,"source":"arxiv_source","source_observed_at":"2026-08-15T20:21:22.143383Z"},"links":{"cited_paper":"/paper/2308.03688","citing_paper":"/paper/2505.13328"},"observation_digest":"sha256:5bcf454e8dcb60e3488f77ee8e47185cce49e293af471f49a7df3ba7f89e53d8","observation_id":"470239f6-ba28-40e9-bb14-c5994ac86121","resolution":{"observed_at":"2026-08-15T20:21:22.143383Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2408.04682","last_updated":"2025-04-16T22:20:21Z","snapshot_observed_at":"2026-08-16T13:27:47.353876Z","submitted_at":"2024-08-08T05:45:42Z","title":"ToolSandbox: A Stateful, Conversational, Interactive Evaluation Benchmark for LLM Tool Use Capabilities","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.04682","snapshot_observed_at":"2026-08-15T20:21:22.209839Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.13328","last_updated":"2025-05-19T16:36:13Z","snapshot_observed_at":"2026-08-23T02:31:24.768215Z","submitted_at":"2025-05-19T16:36:13Z","title":"Rethinking Stateful Tool Use in Multi-Turn Dialogues: Benchmarks and Challenges","version":1},"reference_index":20,"source":"arxiv_source","source_observed_at":"2026-08-15T20:21:22.209839Z"},"links":{"cited_paper":"/paper/2408.04682","citing_paper":"/paper/2505.13328"},"observation_digest":"sha256:a4335348ce5792dadb4ad800fa823de9a3e83eaaddae2d9d386d45ef4aaeab18","observation_id":"1873e3ea-ba02-4f5c-a9c5-b8504ac0da62","resolution":{"observed_at":"2026-08-15T20:21:22.209839Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.09842","last_updated":"2023-10-31T17:43:39Z","snapshot_observed_at":"2026-08-18T03:05:32.170128Z","submitted_at":"2023-04-19T17:47:47Z","title":"Chameleon: Plug-and-Play Compositional Reasoning with Large Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.09842","snapshot_observed_at":"2026-08-15T20:21:22.234517Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.13328","last_updated":"2025-05-19T16:36:13Z","snapshot_observed_at":"2026-08-23T02:31:24.768215Z","submitted_at":"2025-05-19T16:36:13Z","title":"Rethinking Stateful Tool Use in Multi-Turn Dialogues: Benchmarks and Challenges","version":1},"reference_index":21,"source":"arxiv_source","source_observed_at":"2026-08-15T20:21:22.234517Z"},"links":{"cited_paper":"/paper/2304.09842","citing_paper":"/paper/2505.13328"},"observation_digest":"sha256:e16d8ea3b7337ed3675bde22b44a9be8258d9cc7f51186f255e88d52f8b24e44","observation_id":"c8329736-4908-4354-9036-50ef0c05b8ce","resolution":{"observed_at":"2026-08-15T20:21:22.234517Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.13178","last_updated":"2024-12-23T20:12:48Z","snapshot_observed_at":"2026-08-20T13:31:05.293801Z","submitted_at":"2024-01-24T01:51:00Z","title":"AgentBoard: An Analytical Evaluation Board of Multi-turn LLM Agents","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.13178","snapshot_observed_at":"2026-08-15T20:21:22.239300Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.13328","last_updated":"2025-05-19T16:36:13Z","snapshot_observed_at":"2026-08-23T02:31:24.768215Z","submitted_at":"2025-05-19T16:36:13Z","title":"Rethinking Stateful Tool Use in Multi-Turn Dialogues: Benchmarks and Challenges","version":1},"reference_index":22,"source":"arxiv_source","source_observed_at":"2026-08-15T20:21:22.239300Z"},"links":{"cited_paper":"/paper/2401.13178","citing_paper":"/paper/2505.13328"},"observation_digest":"sha256:c41a41bbb5a7513b36cb6edd10178fd18d7f2b72d10ecafbbb939ddfaca8795a","observation_id":"e12d8fcb-21d4-4a51-9490-6da9ccbc8024","resolution":{"observed_at":"2026-08-15T20:21:22.239300Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.12983","last_updated":"2023-11-21T20:34:47Z","snapshot_observed_at":"2026-08-13T10:06:26.439949Z","submitted_at":"2023-11-21T20:34:47Z","title":"GAIA: a benchmark for General AI Assistants","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.12983","snapshot_observed_at":"2026-08-15T20:21:22.243604Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.13328","last_updated":"2025-05-19T16:36:13Z","snapshot_observed_at":"2026-08-23T02:31:24.768215Z","submitted_at":"2025-05-19T16:36:13Z","title":"Rethinking Stateful Tool Use in Multi-Turn Dialogues: Benchmarks and Challenges","version":1},"reference_index":23,"source":"arxiv_source","source_observed_at":"2026-08-15T20:21:22.243604Z"},"links":{"cited_paper":"/paper/2311.12983","citing_paper":"/paper/2505.13328"},"observation_digest":"sha256:ff7842589306ddf42a7357b1d8b37757a732d848278675383edbe444e8859c55","observation_id":"89341ce8-e9a6-4a0e-8d97-5bea7134ae7b","resolution":{"observed_at":"2026-08-15T20:21:22.243604Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2112.09332","last_updated":"2022-06-01T19:08:11Z","snapshot_observed_at":"2026-08-19T02:54:55.334346Z","submitted_at":"2021-12-17T05:43:43Z","title":"WebGPT: Browser-assisted question-answering with human feedback","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2112.09332","snapshot_observed_at":"2026-08-15T20:21:22.248902Z","title":null,"venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2505.13328","last_updated":"2025-05-19T16:36:13Z","snapshot_observed_at":"2026-08-23T02:31:24.768215Z","submitted_at":"2025-05-19T16:36:13Z","title":"Rethinking Stateful Tool Use in Multi-Turn Dialogues: Benchmarks and Challenges","version":1},"reference_index":24,"source":"arxiv_source","source_observed_at":"2026-08-15T20:21:22.248902Z"},"links":{"cited_paper":"/paper/2112.09332","citing_paper":"/paper/2505.13328"},"observation_digest":"sha256:e7ed1125449597607a7fe682e5a1b4aa5bf5894521c6b5ba9d154ba2d948afaa","observation_id":"b2ca764e-9509-47ba-ad41-7913336dff25","resolution":{"observed_at":"2026-08-15T20:21:22.248902Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2305.15334","last_updated":"2023-05-24T16:48:11Z","snapshot_observed_at":"2026-08-20T10:31:15.813764Z","submitted_at":"2023-05-24T16:48:11Z","title":"Gorilla: Large Language Model Connected with Massive APIs","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2305.15334","snapshot_observed_at":"2026-08-15T20:21:22.254047Z","title":"Patil, Tianjun Zhang, Xin Wang, and Joseph E","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.13328","last_updated":"2025-05-19T16:36:13Z","snapshot_observed_at":"2026-08-23T02:31:24.768215Z","submitted_at":"2025-05-19T16:36:13Z","title":"Rethinking Stateful Tool Use in Multi-Turn Dialogues: Benchmarks and Challenges","version":1},"reference_index":25,"source":"arxiv_source","source_observed_at":"2026-08-15T20:21:22.254047Z"},"links":{"cited_paper":"/paper/2305.15334","citing_paper":"/paper/2505.13328"},"observation_digest":"sha256:57c451511d05e343312a9f881c952c6b1b7faa2611d5b3bee739e2c404679dd4","observation_id":"0cff8c72-4ca8-40fe-988c-902fb042ac3d","resolution":{"observed_at":"2026-08-15T20:21:22.254047Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2002.12328","last_updated":"2020-02-27T18:48:33Z","snapshot_observed_at":"2026-08-10T03:54:15.380203Z","submitted_at":"2020-02-27T18:48:33Z","title":"Few-shot Natural Language Generation for Task-Oriented Dialog","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2002.12328","snapshot_observed_at":"2026-08-15T20:21:22.259435Z","title":null,"venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2505.13328","last_updated":"2025-05-19T16:36:13Z","snapshot_observed_at":"2026-08-23T02:31:24.768215Z","submitted_at":"2025-05-19T16:36:13Z","title":"Rethinking Stateful Tool Use in Multi-Turn Dialogues: Benchmarks and Challenges","version":1},"reference_index":26,"source":"arxiv_source","source_observed_at":"2026-08-15T20:21:22.259435Z"},"links":{"cited_paper":"/paper/2002.12328","citing_paper":"/paper/2505.13328"},"observation_digest":"sha256:faa0ce57166271dbc3d91b06b1ca004c3bea7a0300428969b3dca1d9663598e1","observation_id":"e576ddf2-50af-4364-ac8a-b7622f3a0766","resolution":{"observed_at":"2026-08-15T20:21:22.259435Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T20:21:23.381918Z","title":null,"venue":null,"work_id":"3236d42d-8bab-4c6b-a4a8-23ce43fb59ec","year":2018},"citing_paper":{"arxiv_id":"2505.13328","last_updated":"2025-05-19T16:36:13Z","snapshot_observed_at":"2026-08-23T02:31:24.768215Z","submitted_at":"2025-05-19T16:36:13Z","title":"Rethinking Stateful Tool Use in Multi-Turn Dialogues: Benchmarks and Challenges","version":1},"reference_index":27,"source":"arxiv_source","source_observed_at":"2026-08-15T20:21:22.264282Z"},"links":{"citing_paper":"/paper/2505.13328"},"observation_digest":"sha256:20a57555d6c44d2ef78135044c05f3c8c7453e3fe1ef777c983e90a970d6d3a0","observation_id":"7217ea47-1adf-4e73-987c-fb16d574af10","resolution":{"observed_at":"2026-08-15T20:21:23.387003Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T20:21:22.268926Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.13328","last_updated":"2025-05-19T16:36:13Z","snapshot_observed_at":"2026-08-23T02:31:24.768215Z","submitted_at":"2025-05-19T16:36:13Z","title":"Rethinking Stateful Tool Use in Multi-Turn Dialogues: Benchmarks and Challenges","version":1},"reference_index":28,"source":"arxiv_source","source_observed_at":"2026-08-15T20:21:22.268926Z"},"links":{"citing_paper":"/paper/2505.13328"},"observation_digest":"sha256:f048c93e3e4aa0d31a6188f6ac8dec51913fd6b7470f08abfd79844f746c9c73","observation_id":"d0129e7a-bc08-47d9-99b9-e7527832d429","resolution":{"observed_at":"2026-08-15T20:21:22.268926Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2304.08354","last_updated":"2024-08-06T15:14:36Z","snapshot_observed_at":"2026-08-15T21:51:52.440582Z","submitted_at":"2023-04-17T15:16:10Z","title":"Tool Learning with Foundation Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2304.08354","snapshot_observed_at":"2026-08-15T20:21:22.274009Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.13328","last_updated":"2025-05-19T16:36:13Z","snapshot_observed_at":"2026-08-23T02:31:24.768215Z","submitted_at":"2025-05-19T16:36:13Z","title":"Rethinking Stateful Tool Use in Multi-Turn Dialogues: Benchmarks and Challenges","version":1},"reference_index":29,"source":"arxiv_source","source_observed_at":"2026-08-15T20:21:22.274009Z"},"links":{"cited_paper":"/paper/2304.08354","citing_paper":"/paper/2505.13328"},"observation_digest":"sha256:94d0648ed54fbd304028044ba1c6f75e12757f597ecc1b1ca5e67480b183f17e","observation_id":"028a17bb-279e-49fd-90f0-4d6585a659c1","resolution":{"observed_at":"2026-08-15T20:21:22.274009Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2307.16789","last_updated":"2023-10-03T14:45:48Z","snapshot_observed_at":"2026-08-16T04:18:30.717622Z","submitted_at":"2023-07-31T15:56:53Z","title":"ToolLLM: Facilitating Large Language Models to Master 16000+ Real-world APIs","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.16789","snapshot_observed_at":"2026-08-15T20:21:22.279250Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.13328","last_updated":"2025-05-19T16:36:13Z","snapshot_observed_at":"2026-08-23T02:31:24.768215Z","submitted_at":"2025-05-19T16:36:13Z","title":"Rethinking Stateful Tool Use in Multi-Turn Dialogues: Benchmarks and Challenges","version":1},"reference_index":30,"source":"arxiv_source","source_observed_at":"2026-08-15T20:21:22.279250Z"},"links":{"cited_paper":"/paper/2307.16789","citing_paper":"/paper/2505.13328"},"observation_digest":"sha256:81339143fda0d9d9933010a6801b87b3ae8f6389b58f6c17aadfec87fe6f5181","observation_id":"a951ee21-d661-4003-855b-0dcc2f389231","resolution":{"observed_at":"2026-08-15T20:21:22.279250Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T20:21:23.365434Z","title":null,"venue":null,"work_id":"9fba82be-5269-42ed-b3e6-29b37bca2a0d","year":2020},"citing_paper":{"arxiv_id":"2505.13328","last_updated":"2025-05-19T16:36:13Z","snapshot_observed_at":"2026-08-23T02:31:24.768215Z","submitted_at":"2025-05-19T16:36:13Z","title":"Rethinking Stateful Tool Use in Multi-Turn Dialogues: Benchmarks and Challenges","version":1},"reference_index":31,"source":"arxiv_source","source_observed_at":"2026-08-15T20:21:22.284569Z"},"links":{"citing_paper":"/paper/2505.13328"},"observation_digest":"sha256:1721b37f214d474ea4bc2436684de7d2e1ef0604a2ca7cfe75d2859043920822","observation_id":"390fa33f-00f7-476e-b2d5-a0d400046213","resolution":{"observed_at":"2026-08-15T20:21:23.371138Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2303.17580","last_updated":"2023-12-03T18:17:21Z","snapshot_observed_at":"2026-08-12T12:50:51.005571Z","submitted_at":"2023-03-30T17:48:28Z","title":"HuggingGPT: Solving AI Tasks with ChatGPT and its Friends in Hugging Face","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2303.17580","snapshot_observed_at":"2026-08-15T20:21:22.288745Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.13328","last_updated":"2025-05-19T16:36:13Z","snapshot_observed_at":"2026-08-23T02:31:24.768215Z","submitted_at":"2025-05-19T16:36:13Z","title":"Rethinking Stateful Tool Use in Multi-Turn Dialogues: Benchmarks and Challenges","version":1},"reference_index":32,"source":"arxiv_source","source_observed_at":"2026-08-15T20:21:22.288745Z"},"links":{"cited_paper":"/paper/2303.17580","citing_paper":"/paper/2505.13328"},"observation_digest":"sha256:4bab7e79cce1e888976098a34d23a646855692e941c2f33d9a15a15833b2b83f","observation_id":"71c054eb-7db8-4503-af45-66ddefdeec29","resolution":{"observed_at":"2026-08-15T20:21:22.288745Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2212.09946","last_updated":"2022-12-20T01:52:46Z","snapshot_observed_at":"2026-08-18T15:05:53.375844Z","submitted_at":"2022-12-20T01:52:46Z","title":"Dialog2API: Task-Oriented Dialogue with API Description and Example Programs","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2212.09946","snapshot_observed_at":"2026-08-15T20:21:22.293409Z","title":null,"venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2505.13328","last_updated":"2025-05-19T16:36:13Z","snapshot_observed_at":"2026-08-23T02:31:24.768215Z","submitted_at":"2025-05-19T16:36:13Z","title":"Rethinking Stateful Tool Use in Multi-Turn Dialogues: Benchmarks and Challenges","version":1},"reference_index":33,"source":"arxiv_source","source_observed_at":"2026-08-15T20:21:22.293409Z"},"links":{"cited_paper":"/paper/2212.09946","citing_paper":"/paper/2505.13328"},"observation_digest":"sha256:97224c354a007525ccd66b8c0e1ecf4d2fa6861ffa62f6e3975b3fc22ff5970b","observation_id":"a2e76e55-c5e4-47e2-b422-8af1050f183b","resolution":{"observed_at":"2026-08-15T20:21:22.293409Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2309.02427","last_updated":"2024-03-15T15:44:11Z","snapshot_observed_at":"2026-08-18T07:01:35.707347Z","submitted_at":"2023-09-05T17:56:20Z","title":"Cognitive Architectures for Language Agents","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.02427","snapshot_observed_at":"2026-08-15T20:21:22.298603Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.13328","last_updated":"2025-05-19T16:36:13Z","snapshot_observed_at":"2026-08-23T02:31:24.768215Z","submitted_at":"2025-05-19T16:36:13Z","title":"Rethinking Stateful Tool Use in Multi-Turn Dialogues: Benchmarks and Challenges","version":1},"reference_index":34,"source":"arxiv_source","source_observed_at":"2026-08-15T20:21:22.298603Z"},"links":{"cited_paper":"/paper/2309.02427","citing_paper":"/paper/2505.13328"},"observation_digest":"sha256:cc7d223859d4da5e283d2eb650c1ec911219d871f95e4f227039858191a01603","observation_id":"eb7d1599-36f0-43bb-aea7-e5fe910c9b2a","resolution":{"observed_at":"2026-08-15T20:21:22.298603Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.18653/v1/d19-1010","metadata_source":"doi_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T20:21:22.580562Z","title":null,"venue":null,"work_id":"fb93149e-d684-457e-94b5-02da2ab53443","year":2019},"citing_paper":{"arxiv_id":"2505.13328","last_updated":"2025-05-19T16:36:13Z","snapshot_observed_at":"2026-08-23T02:31:24.768215Z","submitted_at":"2025-05-19T16:36:13Z","title":"Rethinking Stateful Tool Use in Multi-Turn Dialogues: Benchmarks and Challenges","version":1},"reference_index":35,"source":"arxiv_source","source_observed_at":"2026-08-15T20:21:22.303475Z"},"links":{"citing_paper":"/paper/2505.13328"},"observation_digest":"sha256:57645a216db8f6aab48c09a9c43cce07c2e190fd071d963fbf256bab487d0ecd","observation_id":"f6d19186-4815-4322-b422-5254acc57821","resolution":{"observed_at":"2026-08-15T20:21:22.586740Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.09288","last_updated":"2023-07-19T17:08:59Z","snapshot_observed_at":"2026-08-07T12:56:43.323460Z","submitted_at":"2023-07-18T14:31:57Z","title":"Llama 2: Open Foundation and Fine-Tuned Chat Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2307.09288","snapshot_observed_at":"2026-08-15T20:21:22.307841Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.13328","last_updated":"2025-05-19T16:36:13Z","snapshot_observed_at":"2026-08-23T02:31:24.768215Z","submitted_at":"2025-05-19T16:36:13Z","title":"Rethinking Stateful Tool Use in Multi-Turn Dialogues: Benchmarks and Challenges","version":1},"reference_index":36,"source":"arxiv_source","source_observed_at":"2026-08-15T20:21:22.307841Z"},"links":{"cited_paper":"/paper/2307.09288","citing_paper":"/paper/2505.13328"},"observation_digest":"sha256:c1381957467e7aadf83a2ac2acfd0b7a78ba2744e863c598dba5f440b456c19d","observation_id":"b5e9d7ac-5cb1-482e-a8ce-676a128a783e","resolution":{"observed_at":"2026-08-15T20:21:22.307841Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.01275","last_updated":"2024-01-09T18:54:05Z","snapshot_observed_at":"2026-08-17T17:53:21.053135Z","submitted_at":"2024-01-02T16:20:40Z","title":"CharacterEval: A Chinese Benchmark for Role-Playing Conversational Agent Evaluation","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.01275","snapshot_observed_at":"2026-08-15T20:21:22.312614Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.13328","last_updated":"2025-05-19T16:36:13Z","snapshot_observed_at":"2026-08-23T02:31:24.768215Z","submitted_at":"2025-05-19T16:36:13Z","title":"Rethinking Stateful Tool Use in Multi-Turn Dialogues: Benchmarks and Challenges","version":1},"reference_index":37,"source":"arxiv_source","source_observed_at":"2026-08-15T20:21:22.312614Z"},"links":{"cited_paper":"/paper/2401.01275","citing_paper":"/paper/2505.13328"},"observation_digest":"sha256:a17350c1a2d27904981aaae3153e9025deba5a7c258af8c8412637faf8b778d8","observation_id":"b7a12258-2f5f-4179-85ae-7ecc4f5a7048","resolution":{"observed_at":"2026-08-15T20:21:22.312614Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.18653/v1/2021.naacl-main.27","metadata_source":"doi_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T20:21:22.561903Z","title":null,"venue":null,"work_id":"3eacb15b-c777-44fa-9719-8475aa63434b","year":2021},"citing_paper":{"arxiv_id":"2505.13328","last_updated":"2025-05-19T16:36:13Z","snapshot_observed_at":"2026-08-23T02:31:24.768215Z","submitted_at":"2025-05-19T16:36:13Z","title":"Rethinking Stateful Tool Use in Multi-Turn Dialogues: Benchmarks and Challenges","version":1},"reference_index":38,"source":"arxiv_source","source_observed_at":"2026-08-15T20:21:22.317286Z"},"links":{"citing_paper":"/paper/2505.13328"},"observation_digest":"sha256:38658a58a5e6b5027162422289c8a18d4e6383ee78c3b76ce1ead86b2044e5b7","observation_id":"deddd860-eb66-406f-8e85-845a06c863db","resolution":{"observed_at":"2026-08-15T20:21:22.569037Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.18653/v1/2023.findings-emnlp.641","metadata_source":"doi_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T20:21:22.540065Z","title":null,"venue":null,"work_id":"1ca267f7-da70-4240-a110-37db56384dd8","year":2023},"citing_paper":{"arxiv_id":"2505.13328","last_updated":"2025-05-19T16:36:13Z","snapshot_observed_at":"2026-08-23T02:31:24.768215Z","submitted_at":"2025-05-19T16:36:13Z","title":"Rethinking Stateful Tool Use in Multi-Turn Dialogues: Benchmarks and Challenges","version":1},"reference_index":39,"source":"arxiv_source","source_observed_at":"2026-08-15T20:21:22.322737Z"},"links":{"citing_paper":"/paper/2505.13328"},"observation_digest":"sha256:52ea340710c315cffc40b009eeeac315490bd8666cf48377cdf74e9334a4cb7c","observation_id":"ab012a2b-fcda-4aec-b796-48417b1d9ad4","resolution":{"observed_at":"2026-08-15T20:21:22.546753Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T20:21:22.327422Z","title":"Pan, and Kam-Fai Wong","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.13328","last_updated":"2025-05-19T16:36:13Z","snapshot_observed_at":"2026-08-23T02:31:24.768215Z","submitted_at":"2025-05-19T16:36:13Z","title":"Rethinking Stateful Tool Use in Multi-Turn Dialogues: Benchmarks and Challenges","version":1},"reference_index":40,"source":"arxiv_source","source_observed_at":"2026-08-15T20:21:22.327422Z"},"links":{"citing_paper":"/paper/2505.13328"},"observation_digest":"sha256:7c7516b0ebd4c32e83fd8dbdce3f90a2c83d6f92bfb740c88741121db31d136d","observation_id":"99db0743-4b5f-4621-b857-4e70b2177cfc","resolution":{"observed_at":"2026-08-15T20:21:22.327422Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T20:21:23.348302Z","title":null,"venue":null,"work_id":"0e549af6-353e-401b-9e35-51c30d50b707","year":2022},"citing_paper":{"arxiv_id":"2505.13328","last_updated":"2025-05-19T16:36:13Z","snapshot_observed_at":"2026-08-23T02:31:24.768215Z","submitted_at":"2025-05-19T16:36:13Z","title":"Rethinking Stateful Tool Use in Multi-Turn Dialogues: Benchmarks and Challenges","version":1},"reference_index":41,"source":"arxiv_source","source_observed_at":"2026-08-15T20:21:22.332150Z"},"links":{"citing_paper":"/paper/2505.13328"},"observation_digest":"sha256:e3c4e441de3ec5e3e55c166cf57768827bdf7ae4c24c58eb52cabcd3099d8f4c","observation_id":"4ceeb9ad-05f4-47ca-8a5b-c565493e841e","resolution":{"observed_at":"2026-08-15T20:21:23.353866Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-23T06:30:58.430688+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.16789","last_updated":"2025-07-20T10:06:23Z","snapshot_observed_at":"2026-08-16T14:39:38.554777Z","submitted_at":"2023-11-28T13:51:32Z","title":"A Survey of the Evolution of Language Model-Based Dialogue Systems: Data, Task and Models","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.16789","snapshot_observed_at":"2026-08-15T20:21:22.337559Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.13328","last_updated":"2025-05-19T16:36:13Z","snapshot_observed_at":"2026-08-23T02:31:24.768215Z","submitted_at":"2025-05-19T16:36:13Z","title":"Rethinking Stateful Tool Use in Multi-Turn Dialogues: Benchmarks and Challenges","version":1},"reference_index":42,"source":"arxiv_source","source_observed_at":"2026-08-15T20:21:22.337559Z"},"links":{"cited_paper":"/paper/2311.16789","citing_paper":"/paper/2505.13328"},"observation_digest":"sha256:6c0749e7e758f638f18cd8a3e9895c19ef7ba923dc618e8593a8c06333e05e9e","observation_id":"0ecedaed-9c0a-450f-9912-e53c351e171b","resolution":{"observed_at":"2026-08-15T20:21:22.337559Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T20:21:22.342635Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.13328","last_updated":"2025-05-19T16:36:13Z","snapshot_observed_at":"2026-08-23T02:31:24.768215Z","submitted_at":"2025-05-19T16:36:13Z","title":"Rethinking Stateful Tool Use in Multi-Turn Dialogues: Benchmarks and Challenges","version":1},"reference_index":43,"source":"arxiv_source","source_observed_at":"2026-08-15T20:21:22.342635Z"},"links":{"citing_paper":"/paper/2505.13328"},"observation_digest":"sha256:b169bc3c6a8077f809ea3ba76031191462b5bbabc554f7552eafb9f998883555","observation_id":"2e29f278-418c-49e0-ad6f-1b49d6e7804d","resolution":{"observed_at":"2026-08-15T20:21:22.342635Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T20:21:22.348097Z","title":"Pan, and Kam-Fai Wong","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.13328","last_updated":"2025-05-19T16:36:13Z","snapshot_observed_at":"2026-08-23T02:31:24.768215Z","submitted_at":"2025-05-19T16:36:13Z","title":"Rethinking Stateful Tool Use in Multi-Turn Dialogues: Benchmarks and Challenges","version":1},"reference_index":44,"source":"arxiv_source","source_observed_at":"2026-08-15T20:21:22.348097Z"},"links":{"citing_paper":"/paper/2505.13328"},"observation_digest":"sha256:829a3a16500a4779a3b9ef36aecb4ae48b45b042fdb66d0c06fcbcb502171c3b","observation_id":"81ecf18d-2150-41a5-be36-529c63b84e47","resolution":{"observed_at":"2026-08-15T20:21:22.348097Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T20:21:22.353417Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.13328","last_updated":"2025-05-19T16:36:13Z","snapshot_observed_at":"2026-08-23T02:31:24.768215Z","submitted_at":"2025-05-19T16:36:13Z","title":"Rethinking Stateful Tool Use in Multi-Turn Dialogues: Benchmarks and Challenges","version":1},"reference_index":45,"source":"arxiv_source","source_observed_at":"2026-08-15T20:21:22.353417Z"},"links":{"citing_paper":"/paper/2505.13328"},"observation_digest":"sha256:aebc59e21af0a5d2a78802c3ab9dcb43f3f66a16e5e5e22d67e9a4014ba7cc3a","observation_id":"759242c7-8a97-4167-885d-c296addf5fa3","resolution":{"observed_at":"2026-08-15T20:21:22.353417Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T20:21:22.357700Z","title":null,"venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2505.13328","last_updated":"2025-05-19T16:36:13Z","snapshot_observed_at":"2026-08-23T02:31:24.768215Z","submitted_at":"2025-05-19T16:36:13Z","title":"Rethinking Stateful Tool Use in Multi-Turn Dialogues: Benchmarks and Challenges","version":1},"reference_index":46,"source":"arxiv_source","source_observed_at":"2026-08-15T20:21:22.357700Z"},"links":{"citing_paper":"/paper/2505.13328"},"observation_digest":"sha256:e3b6f404b5561d16a7b78df134dd0b903587ac6dba1f22bd7cffb0806c9ca08f","observation_id":"2c9f7a41-f11c-4b04-9037-dea4e97f8840","resolution":{"observed_at":"2026-08-15T20:21:22.357700Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2309.10691","last_updated":"2024-03-12T15:53:06Z","snapshot_observed_at":"2026-08-18T10:42:25.094312Z","submitted_at":"2023-09-19T15:25:42Z","title":"MINT: Evaluating LLMs in Multi-turn Interaction with Tools and Language Feedback","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.10691","snapshot_observed_at":"2026-08-15T20:21:22.362158Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.13328","last_updated":"2025-05-19T16:36:13Z","snapshot_observed_at":"2026-08-23T02:31:24.768215Z","submitted_at":"2025-05-19T16:36:13Z","title":"Rethinking Stateful Tool Use in Multi-Turn Dialogues: Benchmarks and Challenges","version":1},"reference_index":47,"source":"arxiv_source","source_observed_at":"2026-08-15T20:21:22.362158Z"},"links":{"cited_paper":"/paper/2309.10691","citing_paper":"/paper/2505.13328"},"observation_digest":"sha256:bbe87fea97016823b15d4e5d7a226fded4dc952113cfbb1c247e1b64f6013ddb","observation_id":"2099b230-e2c8-494b-8a8f-8fc1a85b6b63","resolution":{"observed_at":"2026-08-15T20:21:22.362158Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.00746","last_updated":"2024-06-18T13:08:24Z","snapshot_observed_at":"2026-08-19T12:40:58.780256Z","submitted_at":"2023-10-01T17:52:59Z","title":"RoleLLM: Benchmarking, Eliciting, and Enhancing Role-Playing Abilities of Large Language Models","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.00746","snapshot_observed_at":"2026-08-15T20:21:22.366756Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.13328","last_updated":"2025-05-19T16:36:13Z","snapshot_observed_at":"2026-08-23T02:31:24.768215Z","submitted_at":"2025-05-19T16:36:13Z","title":"Rethinking Stateful Tool Use in Multi-Turn Dialogues: Benchmarks and Challenges","version":1},"reference_index":48,"source":"arxiv_source","source_observed_at":"2026-08-15T20:21:22.366756Z"},"links":{"cited_paper":"/paper/2310.00746","citing_paper":"/paper/2505.13328"},"observation_digest":"sha256:3e49b23718f463616d7c451fde9c600e81008d8095034fb5d69316ec88511355","observation_id":"f4916298-04a0-4710-b894-8bac46be5fef","resolution":{"observed_at":"2026-08-15T20:21:22.366756Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T20:21:22.371409Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.13328","last_updated":"2025-05-19T16:36:13Z","snapshot_observed_at":"2026-08-23T02:31:24.768215Z","submitted_at":"2025-05-19T16:36:13Z","title":"Rethinking Stateful Tool Use in Multi-Turn Dialogues: Benchmarks and Challenges","version":1},"reference_index":49,"source":"arxiv_source","source_observed_at":"2026-08-15T20:21:22.371409Z"},"links":{"citing_paper":"/paper/2505.13328"},"observation_digest":"sha256:55055a21a30cd2ea5ff1972cdbebf54a11a95abdfc7c0078305f60b43c6301e1","observation_id":"e7c7d5a7-d410-46cd-b7f3-337db7b24c12","resolution":{"observed_at":"2026-08-15T20:21:22.371409Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T20:21:22.376398Z","title":null,"venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2505.13328","last_updated":"2025-05-19T16:36:13Z","snapshot_observed_at":"2026-08-23T02:31:24.768215Z","submitted_at":"2025-05-19T16:36:13Z","title":"Rethinking Stateful Tool Use in Multi-Turn Dialogues: Benchmarks and Challenges","version":1},"reference_index":50,"source":"arxiv_source","source_observed_at":"2026-08-15T20:21:22.376398Z"},"links":{"citing_paper":"/paper/2505.13328"},"observation_digest":"sha256:63a96853f9c60a4b6be8fdf606641c3de73a31ac2fef9d2655cdf45f0aa1fb4a","observation_id":"c8f4ca26-2ce3-461f-8aaf-fc19f807f188","resolution":{"observed_at":"2026-08-15T20:21:22.376398Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.12045","last_updated":"2024-06-17T19:33:08Z","snapshot_observed_at":"2026-08-17T20:31:29.818313Z","submitted_at":"2024-06-17T19:33:08Z","title":"$\\tau$-bench: A Benchmark for Tool-Agent-User Interaction in Real-World Domains","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.12045","snapshot_observed_at":"2026-08-15T20:21:22.381083Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.13328","last_updated":"2025-05-19T16:36:13Z","snapshot_observed_at":"2026-08-23T02:31:24.768215Z","submitted_at":"2025-05-19T16:36:13Z","title":"Rethinking Stateful Tool Use in Multi-Turn Dialogues: Benchmarks and Challenges","version":1},"reference_index":51,"source":"arxiv_source","source_observed_at":"2026-08-15T20:21:22.381083Z"},"links":{"cited_paper":"/paper/2406.12045","citing_paper":"/paper/2505.13328"},"observation_digest":"sha256:085ecc6d39ef346b20a2c2fcd97521cc271b873ff2a2fc7f912490dbb58e0b6e","observation_id":"5e6b3a31-4483-40ae-9fa7-d7be07bfd267","resolution":{"observed_at":"2026-08-15T20:21:22.381083Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T20:21:22.385935Z","title":null,"venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2505.13328","last_updated":"2025-05-19T16:36:13Z","snapshot_observed_at":"2026-08-23T02:31:24.768215Z","submitted_at":"2025-05-19T16:36:13Z","title":"Rethinking Stateful Tool Use in Multi-Turn Dialogues: Benchmarks and Challenges","version":1},"reference_index":52,"source":"arxiv_source","source_observed_at":"2026-08-15T20:21:22.385935Z"},"links":{"citing_paper":"/paper/2505.13328"},"observation_digest":"sha256:aac58b747fcf73d128e6cc70e2822e301c38c695164d72ae2c9995b1cba50d61","observation_id":"5faf31fe-b4b0-4e01-a865-5d3f436bd377","resolution":{"observed_at":"2026-08-15T20:21:22.385935Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T20:21:22.390766Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.13328","last_updated":"2025-05-19T16:36:13Z","snapshot_observed_at":"2026-08-23T02:31:24.768215Z","submitted_at":"2025-05-19T16:36:13Z","title":"Rethinking Stateful Tool Use in Multi-Turn Dialogues: Benchmarks and Challenges","version":1},"reference_index":53,"source":"arxiv_source","source_observed_at":"2026-08-15T20:21:22.390766Z"},"links":{"citing_paper":"/paper/2505.13328"},"observation_digest":"sha256:21353f9d6d7c57696f8d4232e86d65f2c6522356405272262b41cf0865a7dfcc","observation_id":"cab0b35d-329d-45fa-9469-cd064975dac4","resolution":{"observed_at":"2026-08-15T20:21:22.390766Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.16832","last_updated":"2023-11-28T14:49:23Z","snapshot_observed_at":"2026-08-20T01:51:25.694111Z","submitted_at":"2023-11-28T14:49:23Z","title":"CharacterGLM: Customizing Chinese Conversational AI Characters with Large Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.16832","snapshot_observed_at":"2026-08-15T20:21:22.395180Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.13328","last_updated":"2025-05-19T16:36:13Z","snapshot_observed_at":"2026-08-23T02:31:24.768215Z","submitted_at":"2025-05-19T16:36:13Z","title":"Rethinking Stateful Tool Use in Multi-Turn Dialogues: Benchmarks and Challenges","version":1},"reference_index":54,"source":"arxiv_source","source_observed_at":"2026-08-15T20:21:22.395180Z"},"links":{"cited_paper":"/paper/2311.16832","citing_paper":"/paper/2505.13328"},"observation_digest":"sha256:5eaa48181e025661dc579da1372d9bb178290ed4b4b846d8a02397d1deb3178e","observation_id":"6ea6efc9-9cc3-4485-bab7-e3c71d23b362","resolution":{"observed_at":"2026-08-15T20:21:22.395180Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T20:21:22.400298Z","title":null,"venue":null,"work_id":null,"year":2020},"citing_paper":{"arxiv_id":"2505.13328","last_updated":"2025-05-19T16:36:13Z","snapshot_observed_at":"2026-08-23T02:31:24.768215Z","submitted_at":"2025-05-19T16:36:13Z","title":"Rethinking Stateful Tool Use in Multi-Turn Dialogues: Benchmarks and Challenges","version":1},"reference_index":55,"source":"arxiv_source","source_observed_at":"2026-08-15T20:21:22.400298Z"},"links":{"citing_paper":"/paper/2505.13328"},"observation_digest":"sha256:b8c7e4afb2974ba37606febb38080b1e804fae971f4b67278b15a3e93ae37d34","observation_id":"e1490c7d-4df1-40b7-b01f-4a1550928802","resolution":{"observed_at":"2026-08-15T20:21:22.400298Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2306.13304","last_updated":"2023-06-23T05:43:28Z","snapshot_observed_at":"2026-08-19T05:45:53.973699Z","submitted_at":"2023-06-23T05:43:28Z","title":"ToolQA: A Dataset for LLM Question Answering with External Tools","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2306.13304","snapshot_observed_at":"2026-08-15T20:21:22.405524Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.13328","last_updated":"2025-05-19T16:36:13Z","snapshot_observed_at":"2026-08-23T02:31:24.768215Z","submitted_at":"2025-05-19T16:36:13Z","title":"Rethinking Stateful Tool Use in Multi-Turn Dialogues: Benchmarks and Challenges","version":1},"reference_index":56,"source":"arxiv_source","source_observed_at":"2026-08-15T20:21:22.405524Z"},"links":{"cited_paper":"/paper/2306.13304","citing_paper":"/paper/2505.13328"},"observation_digest":"sha256:5e24203ad58f86f02c5745b1bbcbb56d2e116e8eb74ea8b94d639c78cc830c55","observation_id":"adc348c2-4bb4-48c2-873d-02333adc2639","resolution":{"observed_at":"2026-08-15T20:21:22.405524Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T20:21:22.410440Z","title":"online\" 'onlinestring :=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.13328","last_updated":"2025-05-19T16:36:13Z","snapshot_observed_at":"2026-08-23T02:31:24.768215Z","submitted_at":"2025-05-19T16:36:13Z","title":"Rethinking Stateful Tool Use in Multi-Turn Dialogues: Benchmarks and Challenges","version":1},"reference_index":57,"source":"arxiv_source","source_observed_at":"2026-08-15T20:21:22.410440Z"},"links":{"citing_paper":"/paper/2505.13328"},"observation_digest":"sha256:dc4fc81e01e70c10b4cb92b739d2aabb0860375b3a267808343ab80baaf1bf65","observation_id":"89db7830-05bd-45a3-bcf7-d1af6c9e01eb","resolution":{"observed_at":"2026-08-15T20:21:22.410440Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-15T20:21:22.416314Z","title":"write newline","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.13328","last_updated":"2025-05-19T16:36:13Z","snapshot_observed_at":"2026-08-23T02:31:24.768215Z","submitted_at":"2025-05-19T16:36:13Z","title":"Rethinking Stateful Tool Use in Multi-Turn Dialogues: Benchmarks and Challenges","version":1},"reference_index":58,"source":"arxiv_source","source_observed_at":"2026-08-15T20:21:22.416314Z"},"links":{"citing_paper":"/paper/2505.13328"},"observation_digest":"sha256:453f9a3c9c33d582f7373e58fceecb4ee3b825eb2998ba46ec0dca114f470909","observation_id":"0776f5d2-6d41-49b7-8761-ea38604b646b","resolution":{"observed_at":"2026-08-15T20:21:22.416314Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"paper":{"arxiv_id":"2505.13328","last_updated":"2025-05-19T16:36:13Z","latest_version":1,"primary_category":"cs.CL","snapshot_observed_at":"2026-08-23T02:31:24.768215Z","submitted_at":"2025-05-19T16:36:13Z","title":"Rethinking Stateful Tool Use in Multi-Turn Dialogues: Benchmarks and Challenges"},"reference_resolution":{"displayed":58,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":55,"verified_exact":3,"verified_fuzzy":0},"total_outbound_references":58},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-23T06:30:58.430688+00:00","source":"crossref"},{"observed_at":"2026-08-23T06:30:53.778098+00:00","source":"retraction_watch"}],"thesis":"As of 23 August 2026, this Paper Citation Record lists 58 of 58 outbound references and 2 inbound Pith citation observations for arXiv:2505.13328."}