{"as_of":"2026-08-13T07:29:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:60c7a4dd6c6ba99b72f5cdbf5343811264835b212d59ae74d551d5c9824e4cc6","coverage":[{"denominator":54,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":54,"source":"paper_references, paper_reference_links","source_observed_at":"2026-05-07T05:52:13.100867Z","state":"measured"},{"denominator":63,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":63,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-13T06:32:02.005865+00:00","state":"measured"},{"denominator":9,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":9,"source":"paper_references, paper_reference_links","source_observed_at":"2026-07-31T19:49:39.628527Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"pith","source_observed_at":"2026-07-11T01:37:41.863844Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2604.28139","last_updated":"2026-05-01T09:39:37Z","snapshot_observed_at":"2026-08-11T17:16:41.573956Z","submitted_at":"2026-04-30T17:23:19Z","title":"Claw-Eval-Live: A Live Agent Benchmark for Evolving Real-World Workflows","version":2},"cited_work":{"arxiv_id":"2604.28139","doi":"10.18653/v1/2023.emnlp-demo.28.https://aclanthology.org/2023.emnlp-demo.28/","metadata_source":"pith","pith_arxiv_id":"2604.28139","snapshot_observed_at":"2026-07-11T11:50:26.030339Z","title":"Claw-Eval-Live: A Live Agent Benchmark for Evolving Real-World Workflows","venue":"cs.SE","work_id":"519230b1-eb35-4f2c-a0ed-824ba45c6abd","year":2026},"citing_paper":{"arxiv_id":"2605.10779","last_updated":"2026-05-11T16:14:04Z","snapshot_observed_at":"2026-08-11T10:07:47.669569Z","submitted_at":"2026-05-11T16:14:04Z","title":"LITMUS: Benchmarking Behavioral Jailbreaks of LLM Agents in Real OS Environments","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-05-12T04:08:05.893904Z"},"links":{"cited_paper":"/paper/2604.28139","citing_paper":"/paper/2605.10779"},"observation_digest":"sha256:bdc044b64d6589dc64316bb865a3226680f6cd305d4089fd7015341103a5ac46","observation_id":"2db58369-1310-4a60-83d4-3cbf9dc76cbc","resolution":{"observed_at":"2026-05-12T06:36:26.733753Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2604.28139","last_updated":"2026-05-01T09:39:37Z","snapshot_observed_at":"2026-08-11T17:16:41.573956Z","submitted_at":"2026-04-30T17:23:19Z","title":"Claw-Eval-Live: A Live Agent Benchmark for Evolving Real-World Workflows","version":2},"cited_work":{"arxiv_id":"2604.28139","doi":"10.18653/v1/2023.emnlp-demo.28.https://aclanthology.org/2023.emnlp-demo.28/","metadata_source":"pith","pith_arxiv_id":"2604.28139","snapshot_observed_at":"2026-07-11T11:50:26.030339Z","title":"Claw-Eval-Live: A Live Agent Benchmark for Evolving Real-World Workflows","venue":"cs.SE","work_id":"519230b1-eb35-4f2c-a0ed-824ba45c6abd","year":2026},"citing_paper":{"arxiv_id":"2606.11909","last_updated":"2026-06-10T10:37:27Z","snapshot_observed_at":"2026-08-11T19:19:03.957130Z","submitted_at":"2026-06-10T10:37:27Z","title":"Embodied-BenchClaw: An Autonomous Multi-Agent System for Embodied Spatial Intelligence Benchmark Construction","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-06-27T09:48:49.786021Z"},"links":{"cited_paper":"/paper/2604.28139","citing_paper":"/paper/2606.11909"},"observation_digest":"sha256:34c647c8e3f5b28b95672b387443f74a0c739e1b951e3c658b48ffa3ad85d7cc","observation_id":"487ac342-b28f-494c-abd2-043e589ff16a","resolution":{"observed_at":"2026-07-03T10:48:02.897544Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2604.28139","last_updated":"2026-05-01T09:39:37Z","snapshot_observed_at":"2026-08-11T17:16:41.573956Z","submitted_at":"2026-04-30T17:23:19Z","title":"Claw-Eval-Live: A Live Agent Benchmark for Evolving Real-World Workflows","version":2},"cited_work":{"arxiv_id":"2604.28139","doi":"10.18653/v1/2023.emnlp-demo.28.https://aclanthology.org/2023.emnlp-demo.28/","metadata_source":"pith","pith_arxiv_id":"2604.28139","snapshot_observed_at":"2026-07-11T11:50:26.030339Z","title":"Claw-Eval-Live: A Live Agent Benchmark for Evolving Real-World Workflows","venue":"cs.SE","work_id":"519230b1-eb35-4f2c-a0ed-824ba45c6abd","year":2026},"citing_paper":{"arxiv_id":"2606.23049","last_updated":"2026-06-24T02:20:35Z","snapshot_observed_at":"2026-08-12T15:24:56.726828Z","submitted_at":"2026-06-22T08:57:54Z","title":"PhoneBuddy: Training Open Models for Agentic Phone Use","version":2},"reference_index":72,"source":"arxiv_source","source_observed_at":"2026-06-26T08:15:49.428124Z"},"links":{"cited_paper":"/paper/2604.28139","citing_paper":"/paper/2606.23049"},"observation_digest":"sha256:5fff5104f8d8532a3bca7d0ac865dc794ea93c79692c3e7b1c3b639208642b65","observation_id":"bf8f3d28-7526-457b-a41d-d0c47d8c2aca","resolution":{"observed_at":"2026-07-04T10:59:46.390185Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2604.28139","last_updated":"2026-05-01T09:39:37Z","snapshot_observed_at":"2026-08-11T17:16:41.573956Z","submitted_at":"2026-04-30T17:23:19Z","title":"Claw-Eval-Live: A Live Agent Benchmark for Evolving Real-World Workflows","version":2},"cited_work":{"arxiv_id":"2604.28139","doi":"10.18653/v1/2023.emnlp-demo.28.https://aclanthology.org/2023.emnlp-demo.28/","metadata_source":"pith","pith_arxiv_id":"2604.28139","snapshot_observed_at":"2026-07-11T11:50:26.030339Z","title":"Claw-Eval-Live: A Live Agent Benchmark for Evolving Real-World Workflows","venue":"cs.SE","work_id":"519230b1-eb35-4f2c-a0ed-824ba45c6abd","year":2026},"citing_paper":{"arxiv_id":"2607.06008","last_updated":"2026-07-09T11:01:43Z","snapshot_observed_at":"2026-08-10T00:40:31.334475Z","submitted_at":"2026-07-07T08:50:09Z","title":"PolyWorkBench: Benchmarking Multilingual Long-Horizon LLM Agents","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-07-11T11:50:26.030339Z"},"links":{"cited_paper":"/paper/2604.28139","citing_paper":"/paper/2607.06008"},"observation_digest":"sha256:a2bd3fce52db71b460403b926497466d8c02d8f3a54fc461bf7f06b93d6999f6","observation_id":"1941e47a-9ccf-474a-a68d-fd83b4d57e76","resolution":{"observed_at":"2026-07-08T20:05:33.706856Z","resolver_source":"local_arxiv","status":"malformed_identifier"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2604.28139","last_updated":"2026-05-01T09:39:37Z","snapshot_observed_at":"2026-08-11T17:16:41.573956Z","submitted_at":"2026-04-30T17:23:19Z","title":"Claw-Eval-Live: A Live Agent Benchmark for Evolving Real-World Workflows","version":2},"cited_work":{"arxiv_id":"2604.28139","doi":"10.18653/v1/2023.emnlp-demo.28.https://aclanthology.org/2023.emnlp-demo.28/","metadata_source":"pith","pith_arxiv_id":"2604.28139","snapshot_observed_at":"2026-07-11T11:50:26.030339Z","title":"Claw-Eval-Live: A Live Agent Benchmark for Evolving Real-World Workflows","venue":"cs.SE","work_id":"519230b1-eb35-4f2c-a0ed-824ba45c6abd","year":2026},"citing_paper":{"arxiv_id":"2607.06008","last_updated":"2026-07-09T11:01:43Z","snapshot_observed_at":"2026-08-10T00:40:31.334475Z","submitted_at":"2026-07-07T08:50:09Z","title":"PolyWorkBench: Benchmarking Multilingual Long-Horizon LLM Agents","version":2},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-07-11T11:50:26.030339Z"},"links":{"cited_paper":"/paper/2604.28139","citing_paper":"/paper/2607.06008"},"observation_digest":"sha256:9df07a63d7c28f29d1cab07513c7fa97fd37446b11dd7c30f9706d4c6a3eb92a","observation_id":"71407461-773a-4b33-add0-da937d2de379","resolution":{"observed_at":"2026-07-11T01:37:41.892647Z","resolver_source":"local_arxiv","status":"malformed_identifier"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2604.28139","last_updated":"2026-05-01T09:39:37Z","snapshot_observed_at":"2026-08-11T17:16:41.573956Z","submitted_at":"2026-04-30T17:23:19Z","title":"Claw-Eval-Live: A Live Agent Benchmark for Evolving Real-World Workflows","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2604.28139","snapshot_observed_at":"2026-07-30T20:14:27.105709Z","title":"Claw-eval-live: A live agent benchmark for evolving real-world workflows, 2026","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2607.23524","last_updated":"2026-07-26T07:45:01Z","snapshot_observed_at":"2026-08-07T06:30:09.161251Z","submitted_at":"2026-07-26T07:45:01Z","title":"Delegation Intelligence in Deep Search: A Controllable Framework for Disentangled Capability Diagnosis","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-07-30T20:14:27.105709Z"},"links":{"cited_paper":"/paper/2604.28139","citing_paper":"/paper/2607.23524"},"observation_digest":"sha256:d71ba89ebdbec35ba9ddb5c9e5afd15e0ac5bc5998dfbbcbd7ae386c00d25980","observation_id":"3808e870-7df5-4940-ba46-6139be23fd5c","resolution":{"observed_at":"2026-07-30T20:14:27.105709Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2604.28139","last_updated":"2026-05-01T09:39:37Z","snapshot_observed_at":"2026-08-11T17:16:41.573956Z","submitted_at":"2026-04-30T17:23:19Z","title":"Claw-Eval-Live: A Live Agent Benchmark for Evolving Real-World Workflows","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2604.28139","snapshot_observed_at":"2026-07-30T18:11:58.595774Z","title":"Claw-eval-live: A live agent benchmark for evolving real-world workflows.arXiv preprint arXiv:2604.28139, 2026","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2607.23588","last_updated":"2026-07-26T10:36:24Z","snapshot_observed_at":"2026-08-06T17:14:20.405951Z","submitted_at":"2026-07-26T10:36:24Z","title":"JarvisHub: An Open Harness for Canvas-Native Multimodal Creative Agents","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-07-30T18:11:58.595774Z"},"links":{"cited_paper":"/paper/2604.28139","citing_paper":"/paper/2607.23588"},"observation_digest":"sha256:1ecfc2e74a924822da2d8999b7e88c481c61a1f12643414bb74a968411c6e1ec","observation_id":"34531799-0305-48b4-b461-7dcbccfb4e84","resolution":{"observed_at":"2026-07-30T18:11:58.595774Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2604.28139","last_updated":"2026-05-01T09:39:37Z","snapshot_observed_at":"2026-08-11T17:16:41.573956Z","submitted_at":"2026-04-30T17:23:19Z","title":"Claw-Eval-Live: A Live Agent Benchmark for Evolving Real-World Workflows","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2604.28139","snapshot_observed_at":"2026-07-30T21:00:41.689026Z","title":"arXiv preprint arXiv:2604.28139 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.26791","last_updated":"2026-07-29T11:32:23Z","snapshot_observed_at":"2026-08-06T17:43:00.742184Z","submitted_at":"2026-07-29T11:32:23Z","title":"SecRespond: Benchmarking AI Agents for Real-World Post-Compromise Incident Response","version":1},"reference_index":33,"source":"arxiv_source","source_observed_at":"2026-07-30T21:00:41.689026Z"},"links":{"cited_paper":"/paper/2604.28139","citing_paper":"/paper/2607.26791"},"observation_digest":"sha256:6d4caf08eb4cbad1bd6fe0eae1da50b2cd98120ce2cf57fce265bb03d2257e48","observation_id":"5672d8bc-615b-429a-acef-9de952255485","resolution":{"observed_at":"2026-07-30T21:00:41.689026Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2604.28139","last_updated":"2026-05-01T09:39:37Z","snapshot_observed_at":"2026-08-11T17:16:41.573956Z","submitted_at":"2026-04-30T17:23:19Z","title":"Claw-Eval-Live: A Live Agent Benchmark for Evolving Real-World Workflows","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2604.28139","snapshot_observed_at":"2026-07-31T19:49:39.628527Z","title":null,"venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2607.28037","last_updated":"2026-07-30T11:18:47Z","snapshot_observed_at":"2026-08-10T16:52:16.063664Z","submitted_at":"2026-07-30T11:18:47Z","title":"ClawTrack: Towards Trace-Level Evaluation and Improvement of Real-World Autonomous Agents","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-07-31T19:49:39.628527Z"},"links":{"cited_paper":"/paper/2604.28139","citing_paper":"/paper/2607.28037"},"observation_digest":"sha256:86202082c1bdb0f82f282c874bdf29e194688e9c5698a44c4e18e96b6c533086","observation_id":"976397a2-57d9-4481-8a45-09958c60e8a2","resolution":{"observed_at":"2026-07-31T19:49:39.628527Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2604.28139/citation-record","integrity":"/paper/2604.28139/integrity","json":"/paper/2604.28139/citation-record.json","paper":"/paper/2604.28139"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Claude code.https://www.anthropic.com/product/claude-code","venue":null,"work_id":"e71b1019-1f41-4bd2-bff2-c45d5a6d8bbf","year":2025},"citing_paper":{"arxiv_id":"2604.28139","last_updated":"2026-05-01T09:39:37Z","snapshot_observed_at":"2026-08-11T17:16:41.573956Z","submitted_at":"2026-04-30T17:23:19Z","title":"Claw-Eval-Live: A Live Agent Benchmark for Evolving Real-World Workflows","version":2},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-07T05:52:13.100867Z"},"links":{"citing_paper":"/paper/2604.28139"},"observation_digest":"sha256:e8758c8bd2bdc605a879a6afb9e8db62babd0360399da40ac446a4283edd4070","observation_id":"829ec972-a3f6-45f9-b0cb-0de5cb9d9f27","resolution":{"observed_at":"2026-05-27T10:28:57.871654Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2108.07732","last_updated":"2021-08-16T03:57:30Z","snapshot_observed_at":"2026-08-02T19:23:53.535075Z","submitted_at":"2021-08-16T03:57:30Z","title":"Program Synthesis with Large Language Models","version":1},"cited_work":{"arxiv_id":"2108.07732","doi":"10.1007/s11390-025-5518-5","metadata_source":"pith","pith_arxiv_id":"2108.07732","snapshot_observed_at":"2026-07-11T11:50:26.030339Z","title":"Program Synthesis with Large Language Models","venue":"cs.PL","work_id":"fd241a05-03b9-4de2-9588-9d77ce176125","year":2021},"citing_paper":{"arxiv_id":"2604.28139","last_updated":"2026-05-01T09:39:37Z","snapshot_observed_at":"2026-08-11T17:16:41.573956Z","submitted_at":"2026-04-30T17:23:19Z","title":"Claw-Eval-Live: A Live Agent Benchmark for Evolving Real-World Workflows","version":2},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-07T05:52:13.100867Z"},"links":{"cited_paper":"/paper/2108.07732","citing_paper":"/paper/2604.28139"},"observation_digest":"sha256:5f890cde7f082dc5d8b6c92be413d2995bb96ea50bfe86de74b5162628ce5506","observation_id":"b1250278-e1b5-47a4-9782-c23c6be7cdeb","resolution":{"observed_at":"2026-05-12T10:31:29.113986Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2107.03374","last_updated":"2021-07-14T17:16:02Z","snapshot_observed_at":"2026-08-08T11:58:24.516369Z","submitted_at":"2021-07-07T17:41:24Z","title":"Evaluating Large Language Models Trained on Code","version":2},"cited_work":{"arxiv_id":"2107.03374","doi":"10.48550/arxiv.2107.03374","metadata_source":"pith","pith_arxiv_id":"2107.03374","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Evaluating Large Language Models Trained on Code","venue":"cs.LG","work_id":"042493e9-b26f-4b4e-bbde-382072ca9b08","year":2021},"citing_paper":{"arxiv_id":"2604.28139","last_updated":"2026-05-01T09:39:37Z","snapshot_observed_at":"2026-08-11T17:16:41.573956Z","submitted_at":"2026-04-30T17:23:19Z","title":"Claw-Eval-Live: A Live Agent Benchmark for Evolving Real-World Workflows","version":2},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-07T05:52:13.100867Z"},"links":{"cited_paper":"/paper/2107.03374","citing_paper":"/paper/2604.28139"},"observation_digest":"sha256:8bc6d1f2b90bf1a45207d54ffe5bc44c960fd2b8b600ce7cc05d460f9940d57c","observation_id":"37580116-f617-4c60-9698-e08e01c5b0a2","resolution":{"observed_at":"2026-05-12T10:31:29.127242Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-01T08:08:23.404839+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-01T08:08:23.404839+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.05467","last_updated":"2025-02-28T16:02:27Z","snapshot_observed_at":"2026-08-13T05:20:59.084759Z","submitted_at":"2024-12-06T23:43:59Z","title":"The BrowserGym Ecosystem for Web Agent Research","version":4},"cited_work":{"arxiv_id":"2412.05467","doi":"10.48550/arxiv.2412.05467","metadata_source":"pith","pith_arxiv_id":"2412.05467","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Xu, Siva Reddy, Quentin Cappart, Graham Neubig, Ruslan Salakhutdinov, Nicolas Chapados, and Alexandre Lacoste","venue":"cs.LG","work_id":"f7dd22e4-8dc0-4a62-a313-cb0712e5d7dc","year":2024},"citing_paper":{"arxiv_id":"2604.28139","last_updated":"2026-05-01T09:39:37Z","snapshot_observed_at":"2026-08-11T17:16:41.573956Z","submitted_at":"2026-04-30T17:23:19Z","title":"Claw-Eval-Live: A Live Agent Benchmark for Evolving Real-World Workflows","version":2},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-07T05:52:13.100867Z"},"links":{"cited_paper":"/paper/2412.05467","citing_paper":"/paper/2604.28139"},"observation_digest":"sha256:7d1e4e93d0bddb04ac627f61d4a7cc3554f089034e19c03df5cba6e189e0fc6e","observation_id":"9074fc4d-973e-4ba7-9544-71188b89709a","resolution":{"observed_at":"2026-05-12T10:31:29.159468Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"30efc55a-a30e-44a7-9bd1-0d9464e10cd0","year":2026},"citing_paper":{"arxiv_id":"2604.28139","last_updated":"2026-05-01T09:39:37Z","snapshot_observed_at":"2026-08-11T17:16:41.573956Z","submitted_at":"2026-04-30T17:23:19Z","title":"Claw-Eval-Live: A Live Agent Benchmark for Evolving Real-World Workflows","version":2},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-05-07T05:52:13.100867Z"},"links":{"citing_paper":"/paper/2604.28139"},"observation_digest":"sha256:7155a2f44164a979f687b4e62a3c23f2903b75d53179f9c29f7eb549a30e40d8","observation_id":"ffae4756-848d-4129-b065-47bbb53e4736","resolution":{"observed_at":"2026-05-27T10:28:57.865536Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.06070","last_updated":"2023-12-09T05:57:46Z","snapshot_observed_at":"2026-07-06T15:40:50.807376Z","submitted_at":"2023-06-09T17:44:31Z","title":"Mind2Web: Towards a Generalist Agent for the Web","version":3},"cited_work":{"arxiv_id":"2306.06070","doi":"10.48550/arxiv.2306.06070","metadata_source":"pith","pith_arxiv_id":"2306.06070","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mind2Web: Towards a Generalist Agent for the Web","venue":"cs.CL","work_id":"e26f5a00-c007-439d-83f6-7900f5687b6b","year":2023},"citing_paper":{"arxiv_id":"2604.28139","last_updated":"2026-05-01T09:39:37Z","snapshot_observed_at":"2026-08-11T17:16:41.573956Z","submitted_at":"2026-04-30T17:23:19Z","title":"Claw-Eval-Live: A Live Agent Benchmark for Evolving Real-World Workflows","version":2},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-05-07T05:52:13.100867Z"},"links":{"cited_paper":"/paper/2306.06070","citing_paper":"/paper/2604.28139"},"observation_digest":"sha256:57d2cbaec7c4570ba9b6bfb29b5a417a9a721b395fb2f705a95656ea68000d36","observation_id":"d47521f9-5aa2-4224-8ec5-bbae55678cb1","resolution":{"observed_at":"2026-05-15T20:05:16.335224Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-07-12T03:19:29.437492+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-12T03:19:29.437492+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"333e5c6a-6114-4472-b4c0-16c3c45d72af","year":2026},"citing_paper":{"arxiv_id":"2604.28139","last_updated":"2026-05-01T09:39:37Z","snapshot_observed_at":"2026-08-11T17:16:41.573956Z","submitted_at":"2026-04-30T17:23:19Z","title":"Claw-Eval-Live: A Live Agent Benchmark for Evolving Real-World Workflows","version":2},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-05-07T05:52:13.100867Z"},"links":{"citing_paper":"/paper/2604.28139"},"observation_digest":"sha256:e2c8e3ff1cfbab8a7344e8524a7535e75a24957965b07525d909b896b5da38e2","observation_id":"7c761749-5b98-411a-977d-fde4647503fe","resolution":{"observed_at":"2026-05-27T10:28:57.851295Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.07718","last_updated":"2024-07-23T06:19:28Z","snapshot_observed_at":"2026-08-02T12:09:24.340284Z","submitted_at":"2024-03-12T14:58:45Z","title":"WorkArena: How Capable Are Web Agents at Solving Common Knowledge Work Tasks?","version":5},"cited_work":{"arxiv_id":"2403.07718","doi":"10.48550/arxiv.2403.07718","metadata_source":"pith","pith_arxiv_id":"2403.07718","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"WorkArena: How Capable Are Web Agents at Solving Common Knowledge Work Tasks?","venue":"cs.LG","work_id":"5ac27d9e-4522-46f8-985e-0e4f73130803","year":2024},"citing_paper":{"arxiv_id":"2604.28139","last_updated":"2026-05-01T09:39:37Z","snapshot_observed_at":"2026-08-11T17:16:41.573956Z","submitted_at":"2026-04-30T17:23:19Z","title":"Claw-Eval-Live: A Live Agent Benchmark for Evolving Real-World Workflows","version":2},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-05-07T05:52:13.100867Z"},"links":{"cited_paper":"/paper/2403.07718","citing_paper":"/paper/2604.28139"},"observation_digest":"sha256:49026c3cf89e0e0ba0f2369714073987ac68926442e780697dc883ede92c3c8c","observation_id":"f0e1df9d-2488-4084-8ee8-737150011cae","resolution":{"observed_at":"2026-05-15T02:48:05.307585Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"El Hattami, M","venue":null,"work_id":"fed878d3-e706-4906-b061-6f114a118204","year":2025},"citing_paper":{"arxiv_id":"2604.28139","last_updated":"2026-05-01T09:39:37Z","snapshot_observed_at":"2026-08-11T17:16:41.573956Z","submitted_at":"2026-04-30T17:23:19Z","title":"Claw-Eval-Live: A Live Agent Benchmark for Evolving Real-World Workflows","version":2},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-05-07T05:52:13.100867Z"},"links":{"citing_paper":"/paper/2604.28139"},"observation_digest":"sha256:e38ca50724f7150bd49378835d0f3c8e8a756b64cbe78faf5ff1347ec1b8495f","observation_id":"b2dd6d35-50bb-444d-94e1-1ff7e6f6dc8e","resolution":{"observed_at":"2026-05-27T10:28:57.833906Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2510.18113","doi":"10.48550/arxiv.2510.18113","metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Ersoy, B","venue":"ArXiv.org","work_id":"99ac7ce0-7dc1-4a0f-8624-15e633582b8b","year":2025},"citing_paper":{"arxiv_id":"2604.28139","last_updated":"2026-05-01T09:39:37Z","snapshot_observed_at":"2026-08-11T17:16:41.573956Z","submitted_at":"2026-04-30T17:23:19Z","title":"Claw-Eval-Live: A Live Agent Benchmark for Evolving Real-World Workflows","version":2},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-05-07T05:52:13.100867Z"},"links":{"citing_paper":"/paper/2604.28139"},"observation_digest":"sha256:55f0378c912c058ceeb82cbee0ee11200504f73570b73df111b375fe7c80bc89","observation_id":"01776af8-a86a-4dde-a263-0ad06987787d","resolution":{"observed_at":"2026-05-12T10:31:29.147403Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2401.03065","last_updated":"2024-01-05T20:53:51Z","snapshot_observed_at":"2026-07-06T17:12:10.787061Z","submitted_at":"2024-01-05T20:53:51Z","title":"CRUXEval: A Benchmark for Code Reasoning, Understanding and Execution","version":1},"cited_work":{"arxiv_id":"2401.03065","doi":"10.48550/arxiv.2401.03065","metadata_source":"pith","pith_arxiv_id":"2401.03065","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"CRUXEval: A Benchmark for Code Reasoning, Understanding and Execution","venue":"cs.SE","work_id":"0daed386-84ea-40ec-bdeb-546f8991fca5","year":2024},"citing_paper":{"arxiv_id":"2604.28139","last_updated":"2026-05-01T09:39:37Z","snapshot_observed_at":"2026-08-11T17:16:41.573956Z","submitted_at":"2026-04-30T17:23:19Z","title":"Claw-Eval-Live: A Live Agent Benchmark for Evolving Real-World Workflows","version":2},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-05-07T05:52:13.100867Z"},"links":{"cited_paper":"/paper/2401.03065","citing_paper":"/paper/2604.28139"},"observation_digest":"sha256:078cdcc34f4bba5e744db69e059e0d35aca9a5206ea5c1e7848ad2a7cb5fc6b7","observation_id":"1bde1579-24e4-4c7f-a019-bcef30f73c46","resolution":{"observed_at":"2026-05-14T20:57:16.443569Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"b97511b5-6d61-4987-a4ca-dcafb3d87119","year":2024},"citing_paper":{"arxiv_id":"2604.28139","last_updated":"2026-05-01T09:39:37Z","snapshot_observed_at":"2026-08-11T17:16:41.573956Z","submitted_at":"2026-04-30T17:23:19Z","title":"Claw-Eval-Live: A Live Agent Benchmark for Evolving Real-World Workflows","version":2},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-05-07T05:52:13.100867Z"},"links":{"citing_paper":"/paper/2604.28139"},"observation_digest":"sha256:31ba5c3119186eda17c8a27b75ecf5bc66d71d984d4e1736ddaf5a89823bae99","observation_id":"d19868e1-b300-4d8f-9600-bc6891799431","resolution":{"observed_at":"2026-05-27T10:28:57.807907Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Researchclawbench: Evaluating ai agents for automated research from re- discovery to new-discovery","venue":null,"work_id":"48797984-8821-4961-8de7-6162db44755d","year":2026},"citing_paper":{"arxiv_id":"2604.28139","last_updated":"2026-05-01T09:39:37Z","snapshot_observed_at":"2026-08-11T17:16:41.573956Z","submitted_at":"2026-04-30T17:23:19Z","title":"Claw-Eval-Live: A Live Agent Benchmark for Evolving Real-World Workflows","version":2},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-05-07T05:52:13.100867Z"},"links":{"citing_paper":"/paper/2604.28139"},"observation_digest":"sha256:2ac1b8b98de254c2d8ca96c0e9fa48a358c52fe57bbd6a547913cb1f98be0410","observation_id":"51335bfe-9cfc-4e15-91c3-eef90f136d69","resolution":{"observed_at":"2026-05-27T10:28:57.827689Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"681c0bf7-a858-4c88-a4ba-15e57f3b61ea","year":2024},"citing_paper":{"arxiv_id":"2604.28139","last_updated":"2026-05-01T09:39:37Z","snapshot_observed_at":"2026-08-11T17:16:41.573956Z","submitted_at":"2026-04-30T17:23:19Z","title":"Claw-Eval-Live: A Live Agent Benchmark for Evolving Real-World Workflows","version":2},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-05-07T05:52:13.100867Z"},"links":{"citing_paper":"/paper/2604.28139"},"observation_digest":"sha256:960461bf8264112a047883fcd8284b5496ab34666576315da580f46d215bdd6b","observation_id":"830923f9-3017-4c01-9b1c-7763ddf89fc2","resolution":{"observed_at":"2026-05-27T10:28:57.819792Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.06770","last_updated":"2024-11-11T23:05:04Z","snapshot_observed_at":"2026-08-12T14:56:35.025839Z","submitted_at":"2023-10-10T16:47:29Z","title":"SWE-bench: Can Language Models Resolve Real-World GitHub Issues?","version":3},"cited_work":{"arxiv_id":"2310.06770","doi":"10.1145/512927.512945","metadata_source":"pith","pith_arxiv_id":"2310.06770","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SWE-bench: Can Language Models Resolve Real-World GitHub Issues?","venue":"cs.CL","work_id":"d0effe15-a689-441a-8e3f-ea35f1c4e4b1","year":2023},"citing_paper":{"arxiv_id":"2604.28139","last_updated":"2026-05-01T09:39:37Z","snapshot_observed_at":"2026-08-11T17:16:41.573956Z","submitted_at":"2026-04-30T17:23:19Z","title":"Claw-Eval-Live: A Live Agent Benchmark for Evolving Real-World Workflows","version":2},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-05-07T05:52:13.100867Z"},"links":{"cited_paper":"/paper/2310.06770","citing_paper":"/paper/2604.28139"},"observation_digest":"sha256:ab91eb232c43de222f584c76f33cd68e1a20cfb8898e4f84cfcc86da1e420722","observation_id":"0a2b846a-0d52-44ff-8769-ce8581541622","resolution":{"observed_at":"2026-05-12T10:31:29.151001Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Pinchbench: Benchmarking LLM models as OpenClaw coding agents","venue":null,"work_id":"851dddae-4b6e-4a21-9732-e0e50ab66e85","year":2026},"citing_paper":{"arxiv_id":"2604.28139","last_updated":"2026-05-01T09:39:37Z","snapshot_observed_at":"2026-08-11T17:16:41.573956Z","submitted_at":"2026-04-30T17:23:19Z","title":"Claw-Eval-Live: A Live Agent Benchmark for Evolving Real-World Workflows","version":2},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-05-07T05:52:13.100867Z"},"links":{"citing_paper":"/paper/2604.28139"},"observation_digest":"sha256:e10b56ed124a38913031c5a004c6897dd5cb52bda994a76e05114f3d2677985d","observation_id":"085bf094-8474-4561-aae8-7fb4039c78b3","resolution":{"observed_at":"2026-05-27T10:23:57.862451Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"899e7ef9-2851-48e0-8b61-3a0ef4a0d893","year":2024},"citing_paper":{"arxiv_id":"2604.28139","last_updated":"2026-05-01T09:39:37Z","snapshot_observed_at":"2026-08-11T17:16:41.573956Z","submitted_at":"2026-04-30T17:23:19Z","title":"Claw-Eval-Live: A Live Agent Benchmark for Evolving Real-World Workflows","version":2},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-05-07T05:52:13.100867Z"},"links":{"citing_paper":"/paper/2604.28139"},"observation_digest":"sha256:db8d3a73246d3e86c934d93b40c6c4c178ddaf2a91a9f0280bca1028c155c909","observation_id":"dba7664c-be70-440c-a835-5b63c351b1a7","resolution":{"observed_at":"2026-05-27T10:23:57.859335Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"99157f51-48da-4db0-ae48-bb683c295cfc","year":2023},"citing_paper":{"arxiv_id":"2604.28139","last_updated":"2026-05-01T09:39:37Z","snapshot_observed_at":"2026-08-11T17:16:41.573956Z","submitted_at":"2026-04-30T17:23:19Z","title":"Claw-Eval-Live: A Live Agent Benchmark for Evolving Real-World Workflows","version":2},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-05-07T05:52:13.100867Z"},"links":{"citing_paper":"/paper/2604.28139"},"observation_digest":"sha256:675bf713f8e7e2ebc8adfb37019dccff95c55066584a5e2650e61afff692ae1f","observation_id":"40297941-d549-4614-ad4f-58b0edc1c247","resolution":{"observed_at":"2026-05-27T10:28:57.880861Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.06703","last_updated":"2026-06-04T09:59:28Z","snapshot_observed_at":"2026-08-12T22:27:27.158503Z","submitted_at":"2024-10-09T09:13:38Z","title":"ST-WebAgentBench: A Benchmark for Evaluating Safety and Trustworthiness in Web Agents","version":7},"cited_work":{"arxiv_id":"2410.06703","doi":"10.48550/arxiv.2410.06703","metadata_source":"pith","pith_arxiv_id":"2410.06703","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"St- webagentbench: A benchmark for evaluating safety and trustworthiness in web agents","venue":"cs.AI","work_id":"02dc69db-8db6-48b1-ac24-4004642dfa94","year":2024},"citing_paper":{"arxiv_id":"2604.28139","last_updated":"2026-05-01T09:39:37Z","snapshot_observed_at":"2026-08-11T17:16:41.573956Z","submitted_at":"2026-04-30T17:23:19Z","title":"Claw-Eval-Live: A Live Agent Benchmark for Evolving Real-World Workflows","version":2},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-05-07T05:52:13.100867Z"},"links":{"cited_paper":"/paper/2410.06703","citing_paper":"/paper/2604.28139"},"observation_digest":"sha256:98e3d2a64ff710a8db41e3f31360b6254d318c849996637280b9639fd7c3ff8b","observation_id":"a270434b-09db-424e-9899-54c7636497a1","resolution":{"observed_at":"2026-06-05T02:16:16.756993Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.08244","last_updated":"2023-10-25T06:54:12Z","snapshot_observed_at":"2026-08-02T00:07:12.855748Z","submitted_at":"2023-04-14T14:05:32Z","title":"API-Bank: A Comprehensive Benchmark for Tool-Augmented LLMs","version":2},"cited_work":{"arxiv_id":"2304.08244","doi":"10.48550/arxiv.2304.08244","metadata_source":"pith","pith_arxiv_id":"2304.08244","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"API-Bank: A Comprehensive Benchmark for Tool-Augmented LLMs","venue":"cs.CL","work_id":"a20d9332-ab34-485c-a060-1ba47cc98930","year":2023},"citing_paper":{"arxiv_id":"2604.28139","last_updated":"2026-05-01T09:39:37Z","snapshot_observed_at":"2026-08-11T17:16:41.573956Z","submitted_at":"2026-04-30T17:23:19Z","title":"Claw-Eval-Live: A Live Agent Benchmark for Evolving Real-World Workflows","version":2},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-05-07T05:52:13.100867Z"},"links":{"cited_paper":"/paper/2304.08244","citing_paper":"/paper/2604.28139"},"observation_digest":"sha256:7ab33821046de30aa39fe958f8b140d5f633791b3aa688696242ec35dc3c5cd7","observation_id":"759fa504-1109-4f66-8091-e3684506dbd9","resolution":{"observed_at":"2026-05-15T20:51:41.275571Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-07-14T18:20:23.914543+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-14T18:20:23.914543+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.11875","last_updated":"2025-03-14T21:01:28Z","snapshot_observed_at":"2026-08-07T17:02:46.063450Z","submitted_at":"2025-03-14T21:01:28Z","title":"The Galactic population of magnetars : a simulation-based inference study","version":1},"cited_work":{"arxiv_id":"2503.11875","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2503.11875","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"37a39d85-3bf2-431c-a541-f2e9bb4492aa","year":2025},"citing_paper":{"arxiv_id":"2604.28139","last_updated":"2026-05-01T09:39:37Z","snapshot_observed_at":"2026-08-11T17:16:41.573956Z","submitted_at":"2026-04-30T17:23:19Z","title":"Claw-Eval-Live: A Live Agent Benchmark for Evolving Real-World Workflows","version":2},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-05-07T05:52:13.100867Z"},"links":{"cited_paper":"/paper/2503.11875","citing_paper":"/paper/2604.28139"},"observation_digest":"sha256:2cc1b6d61bfcd8fcb08e3f99113598be2b67027611cdf05cbba11023fb686762","observation_id":"20d167c7-d9da-40a5-87c0-e9dc009be947","resolution":{"observed_at":"2026-05-12T10:31:29.172060Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.03091","last_updated":"2023-10-04T01:13:49Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-05T17:59:41Z","title":"RepoBench: Benchmarking Repository-Level Code Auto-Completion Systems","version":2},"cited_work":{"arxiv_id":"2306.03091","doi":"10.48550/arxiv.2306.03091","metadata_source":"pith","pith_arxiv_id":"2306.03091","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"RepoBench: Benchmarking Repository-Level Code Auto-Completion Systems","venue":"cs.CL","work_id":"24b24ba5-b21c-43c6-866f-4b874305372e","year":2023},"citing_paper":{"arxiv_id":"2604.28139","last_updated":"2026-05-01T09:39:37Z","snapshot_observed_at":"2026-08-11T17:16:41.573956Z","submitted_at":"2026-04-30T17:23:19Z","title":"Claw-Eval-Live: A Live Agent Benchmark for Evolving Real-World Workflows","version":2},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-05-07T05:52:13.100867Z"},"links":{"cited_paper":"/paper/2306.03091","citing_paper":"/paper/2604.28139"},"observation_digest":"sha256:d37722089cf02f9dde65803b708ceb300a03e3cdcdb95d5ccd1a21893a34644f","observation_id":"a8fd18c4-9fc1-4c0c-9685-c85c8aee7726","resolution":{"observed_at":"2026-05-15T22:29:52.776747Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-05-21T21:53:10.253516+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-21T21:53:10.253516+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"06360696-147c-4d6e-bcd8-4ff3fa642921","year":2024},"citing_paper":{"arxiv_id":"2604.28139","last_updated":"2026-05-01T09:39:37Z","snapshot_observed_at":"2026-08-11T17:16:41.573956Z","submitted_at":"2026-04-30T17:23:19Z","title":"Claw-Eval-Live: A Live Agent Benchmark for Evolving Real-World Workflows","version":2},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-05-07T05:52:13.100867Z"},"links":{"citing_paper":"/paper/2604.28139"},"observation_digest":"sha256:2d2a858c50449b94a943d97ed7f354e0562eec771b47f689e9794a5db4822583","observation_id":"ce721e82-e784-459a-b3f7-4af0f29bc56b","resolution":{"observed_at":"2026-05-27T10:28:57.838365Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"MacDiarmid, B","venue":null,"work_id":"ab30ddc8-ccb6-408c-ab15-8db781496d5a","year":2025},"citing_paper":{"arxiv_id":"2604.28139","last_updated":"2026-05-01T09:39:37Z","snapshot_observed_at":"2026-08-11T17:16:41.573956Z","submitted_at":"2026-04-30T17:23:19Z","title":"Claw-Eval-Live: A Live Agent Benchmark for Evolving Real-World Workflows","version":2},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-05-07T05:52:13.100867Z"},"links":{"citing_paper":"/paper/2604.28139"},"observation_digest":"sha256:20e081d21689d1e726c48f8630b8c63817dd018f5af36e932823ffdc7a9a4257","observation_id":"541a34d0-cdb4-43ed-990b-995807847f00","resolution":{"observed_at":"2026-05-27T10:28:57.842715Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2602.03221","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Merrill, P","venue":null,"work_id":"5a1051be-439c-4f24-b2d2-9c5c41a48e8c","year":2026},"citing_paper":{"arxiv_id":"2604.28139","last_updated":"2026-05-01T09:39:37Z","snapshot_observed_at":"2026-08-11T17:16:41.573956Z","submitted_at":"2026-04-30T17:23:19Z","title":"Claw-Eval-Live: A Live Agent Benchmark for Evolving Real-World Workflows","version":2},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-05-07T05:52:13.100867Z"},"links":{"citing_paper":"/paper/2604.28139"},"observation_digest":"sha256:8bd8bbec42585c26fd19448c58fa30bed24208e6fc17f3dbc662b1fc0989e103","observation_id":"391e357c-42fb-4519-bf72-b97c027a8d12","resolution":{"observed_at":"2026-05-12T10:31:29.100257Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Mialon, C","venue":null,"work_id":"b58e3d6f-259e-4ea2-ad4e-3722cc8415cf","year":2024},"citing_paper":{"arxiv_id":"2604.28139","last_updated":"2026-05-01T09:39:37Z","snapshot_observed_at":"2026-08-11T17:16:41.573956Z","submitted_at":"2026-04-30T17:23:19Z","title":"Claw-Eval-Live: A Live Agent Benchmark for Evolving Real-World Workflows","version":2},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-05-07T05:52:13.100867Z"},"links":{"citing_paper":"/paper/2604.28139"},"observation_digest":"sha256:14ef443b28f0776ad4965671892f287c58b55afe337ea483fddd43e43188c4b1","observation_id":"b5b2c84d-9b87-450e-bd69-9c6fef007105","resolution":{"observed_at":"2026-05-27T10:28:57.876502Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Hermes agent","venue":null,"work_id":"9ba7324f-da31-4a47-8dfe-013b944d5659","year":null},"citing_paper":{"arxiv_id":"2604.28139","last_updated":"2026-05-01T09:39:37Z","snapshot_observed_at":"2026-08-11T17:16:41.573956Z","submitted_at":"2026-04-30T17:23:19Z","title":"Claw-Eval-Live: A Live Agent Benchmark for Evolving Real-World Workflows","version":2},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-05-07T05:52:13.100867Z"},"links":{"citing_paper":"/paper/2604.28139"},"observation_digest":"sha256:877dd358be4a94320633080c12ff4cda76b0ec7e06cb5566bffc04b90ab2b9b4","observation_id":"da470eb9-4d35-4927-8f6c-734c37e82c84","resolution":{"observed_at":"2026-05-27T10:28:57.855396Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Codex.https://openai.com/codex","venue":null,"work_id":"d85f9bc7-7348-4f66-83b4-8815a404ae57","year":2026},"citing_paper":{"arxiv_id":"2604.28139","last_updated":"2026-05-01T09:39:37Z","snapshot_observed_at":"2026-08-11T17:16:41.573956Z","submitted_at":"2026-04-30T17:23:19Z","title":"Claw-Eval-Live: A Live Agent Benchmark for Evolving Real-World Workflows","version":2},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-05-07T05:52:13.100867Z"},"links":{"citing_paper":"/paper/2604.28139"},"observation_digest":"sha256:49c0a8117af244c332cc574b5fc2e40dd67ef13dcf27dbe3745c4e7713307f8a","observation_id":"04d3bb69-0f9d-4559-8ef9-03bdb5e11512","resolution":{"observed_at":"2026-05-27T10:23:57.853249Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Openclaw","venue":null,"work_id":"00e536d8-7080-48af-a9ae-c0786e5c3751","year":2026},"citing_paper":{"arxiv_id":"2604.28139","last_updated":"2026-05-01T09:39:37Z","snapshot_observed_at":"2026-08-11T17:16:41.573956Z","submitted_at":"2026-04-30T17:23:19Z","title":"Claw-Eval-Live: A Live Agent Benchmark for Evolving Real-World Workflows","version":2},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-05-07T05:52:13.100867Z"},"links":{"citing_paper":"/paper/2604.28139"},"observation_digest":"sha256:37f4e9ee6ea396dfbf287b0ecb882da6a73ad298a0b65172a67ad3131fbd66ee","observation_id":"e86bb791-375b-46be-a06b-b20550c05b99","resolution":{"observed_at":"2026-05-27T10:23:57.844525Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.12373","last_updated":"2024-07-16T06:19:21Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-06-18T07:58:33Z","title":"WebCanvas: Benchmarking Web Agents in Online Environments","version":3},"cited_work":{"arxiv_id":"2406.12373","doi":"10.48550/arxiv.2406.12373","metadata_source":"pith","pith_arxiv_id":"2406.12373","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"WebCanvas: Benchmarking Web Agents in Online Environments","venue":"cs.CL","work_id":"b8949e88-00ed-4d70-b935-0761ce59b669","year":2024},"citing_paper":{"arxiv_id":"2604.28139","last_updated":"2026-05-01T09:39:37Z","snapshot_observed_at":"2026-08-11T17:16:41.573956Z","submitted_at":"2026-04-30T17:23:19Z","title":"Claw-Eval-Live: A Live Agent Benchmark for Evolving Real-World Workflows","version":2},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-05-07T05:52:13.100867Z"},"links":{"cited_paper":"/paper/2406.12373","citing_paper":"/paper/2604.28139"},"observation_digest":"sha256:03bbc3ee37881c5d75f0e0b7272b8d6ab6a73b4f84fd6678abf27b70da866da9","observation_id":"72140e84-7ba9-43c2-a3b0-5f1d5d4c75d7","resolution":{"observed_at":"2026-05-20T11:29:15.078502Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2305.15334","last_updated":"2023-05-24T16:48:11Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-05-24T16:48:11Z","title":"Gorilla: Large Language Model Connected with Massive APIs","version":1},"cited_work":{"arxiv_id":"2305.15334","doi":"10.48550/arxiv.2305.15334","metadata_source":"pith","pith_arxiv_id":"2305.15334","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Gorilla: Large Language Model Connected with Massive APIs","venue":"cs.CL","work_id":"126a464a-4a73-495f-b669-de1e44aa8f09","year":2023},"citing_paper":{"arxiv_id":"2604.28139","last_updated":"2026-05-01T09:39:37Z","snapshot_observed_at":"2026-08-11T17:16:41.573956Z","submitted_at":"2026-04-30T17:23:19Z","title":"Claw-Eval-Live: A Live Agent Benchmark for Evolving Real-World Workflows","version":2},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-05-07T05:52:13.100867Z"},"links":{"cited_paper":"/paper/2305.15334","citing_paper":"/paper/2604.28139"},"observation_digest":"sha256:27eb13c84ed64216f1c700cf2fcd1139e454c8cc00c41b4e0e7e7ff4ef26d146","observation_id":"d7128056-cad7-4066-b938-7b924f07b25c","resolution":{"observed_at":"2026-05-12T10:31:29.135394Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.17541","last_updated":"2024-06-20T02:53:20Z","snapshot_observed_at":"2026-08-13T05:14:36.591429Z","submitted_at":"2023-11-29T11:23:42Z","title":"TaskWeaver: A Code-First Agent Framework","version":3},"cited_work":{"arxiv_id":"2311.17541","doi":"10.48550/arxiv.2311.17541","metadata_source":"arxiv_reference","pith_arxiv_id":"2311.17541","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Taskweaver: A code-first agent framework","venue":"arXiv (Cornell University)","work_id":"456c13ec-7e66-4899-b628-5d5a397d5e33","year":2024},"citing_paper":{"arxiv_id":"2604.28139","last_updated":"2026-05-01T09:39:37Z","snapshot_observed_at":"2026-08-11T17:16:41.573956Z","submitted_at":"2026-04-30T17:23:19Z","title":"Claw-Eval-Live: A Live Agent Benchmark for Evolving Real-World Workflows","version":2},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-05-07T05:52:13.100867Z"},"links":{"cited_paper":"/paper/2311.17541","citing_paper":"/paper/2604.28139"},"observation_digest":"sha256:6da119d40a8ab598dc7a7a3a3d51548ec55e5c661c03c37ae492f76b7718e7ab","observation_id":"2043cd3a-f339-4f34-a99f-072f93cdedfd","resolution":{"observed_at":"2026-05-12T10:31:29.118745Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.16789","last_updated":"2023-10-03T14:45:48Z","snapshot_observed_at":"2026-07-06T16:00:46.542753Z","submitted_at":"2023-07-31T15:56:53Z","title":"ToolLLM: Facilitating Large Language Models to Master 16000+ Real-world APIs","version":2},"cited_work":{"arxiv_id":"2307.16789","doi":"10.48550/arxiv.2307.16789","metadata_source":"pith","pith_arxiv_id":"2307.16789","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"ToolLLM: Facilitating Large Language Models to Master 16000+ Real-world APIs","venue":"cs.AI","work_id":"3c555b48-a4d9-42dd-9fdd-0f6018fbe9cb","year":2023},"citing_paper":{"arxiv_id":"2604.28139","last_updated":"2026-05-01T09:39:37Z","snapshot_observed_at":"2026-08-11T17:16:41.573956Z","submitted_at":"2026-04-30T17:23:19Z","title":"Claw-Eval-Live: A Live Agent Benchmark for Evolving Real-World Workflows","version":2},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-05-07T05:52:13.100867Z"},"links":{"cited_paper":"/paper/2307.16789","citing_paper":"/paper/2604.28139"},"observation_digest":"sha256:3651fff46360c744437c6ab7bd72c30a364d98f6e3d9829eb2e709e1824347b0","observation_id":"16fdd019-dbc8-42c8-841a-486bf5caf1fd","resolution":{"observed_at":"2026-05-12T10:31:29.143547Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-07-12T03:19:33.730697+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-12T03:19:33.730697+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.14573","last_updated":"2025-04-06T20:37:50Z","snapshot_observed_at":"2026-08-12T07:38:32.454201Z","submitted_at":"2024-05-23T13:48:54Z","title":"AndroidWorld: A Dynamic Benchmarking Environment for Autonomous Agents","version":5},"cited_work":{"arxiv_id":"2405.14573","doi":"10.48550/arxiv.2405.14573","metadata_source":"pith","pith_arxiv_id":"2405.14573","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"AndroidWorld: A Dynamic Benchmarking Environment for Autonomous Agents","venue":"cs.AI","work_id":"c5116d19-d3d3-40fd-9620-f7489812a9ba","year":2024},"citing_paper":{"arxiv_id":"2604.28139","last_updated":"2026-05-01T09:39:37Z","snapshot_observed_at":"2026-08-11T17:16:41.573956Z","submitted_at":"2026-04-30T17:23:19Z","title":"Claw-Eval-Live: A Live Agent Benchmark for Evolving Real-World Workflows","version":2},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-05-07T05:52:13.100867Z"},"links":{"cited_paper":"/paper/2405.14573","citing_paper":"/paper/2604.28139"},"observation_digest":"sha256:eeef602a658e239d35f4a73e34f295677a6f9e3d1a980d5c2dd127717d782444","observation_id":"b7f35dc2-ebaf-436a-9e39-2fd00f64d1dc","resolution":{"observed_at":"2026-05-13T12:06:13.928391Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2309.15817","last_updated":"2024-05-17T17:17:45Z","snapshot_observed_at":"2026-07-06T16:24:30.545494Z","submitted_at":"2023-09-25T17:08:02Z","title":"Identifying the Risks of LM Agents with an LM-Emulated Sandbox","version":2},"cited_work":{"arxiv_id":"2309.15817","doi":"10.48550/arxiv.2309.15817","metadata_source":"pith","pith_arxiv_id":"2309.15817","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Identifying the Risks of LM Agents with an LM-Emulated Sandbox","venue":"cs.AI","work_id":"3d4c3b66-d749-4939-b1bc-62b10b2ebbb6","year":2023},"citing_paper":{"arxiv_id":"2604.28139","last_updated":"2026-05-01T09:39:37Z","snapshot_observed_at":"2026-08-11T17:16:41.573956Z","submitted_at":"2026-04-30T17:23:19Z","title":"Claw-Eval-Live: A Live Agent Benchmark for Evolving Real-World Workflows","version":2},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-05-07T05:52:13.100867Z"},"links":{"cited_paper":"/paper/2309.15817","citing_paper":"/paper/2604.28139"},"observation_digest":"sha256:ccfd0c7c32b93e3e06468d16fb8e5bd52d583741c71ce87652cce7f43fe82046","observation_id":"d7830615-f71e-4a2a-abaf-7ac74edb13bc","resolution":{"observed_at":"2026-05-12T20:45:24.770470Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Schick, J","venue":null,"work_id":"c04a80da-91d3-4fac-a71b-34fe40e37702","year":2023},"citing_paper":{"arxiv_id":"2604.28139","last_updated":"2026-05-01T09:39:37Z","snapshot_observed_at":"2026-08-11T17:16:41.573956Z","submitted_at":"2026-04-30T17:23:19Z","title":"Claw-Eval-Live: A Live Agent Benchmark for Evolving Real-World Workflows","version":2},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-05-07T05:52:13.100867Z"},"links":{"citing_paper":"/paper/2604.28139"},"observation_digest":"sha256:2f04261d04a8b8d9d348105f6fa892f8b0067044e731252f50518a91f97e52c7","observation_id":"0714f77a-974d-4469-b061-76677e2d3ecd","resolution":{"observed_at":"2026-05-27T10:23:57.835465Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"a095c68d-d7aa-4190-afd8-685ef3aa3428","year":2023},"citing_paper":{"arxiv_id":"2604.28139","last_updated":"2026-05-01T09:39:37Z","snapshot_observed_at":"2026-08-11T17:16:41.573956Z","submitted_at":"2026-04-30T17:23:19Z","title":"Claw-Eval-Live: A Live Agent Benchmark for Evolving Real-World Workflows","version":2},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-05-07T05:52:13.100867Z"},"links":{"citing_paper":"/paper/2604.28139"},"observation_digest":"sha256:a8a7126e81c0691388187b395d28929fd196b38618634613b1f4f0211f929a5c","observation_id":"e4580784-e309-41b7-a452-5c6a407cc61b","resolution":{"observed_at":"2026-05-27T10:23:57.838199Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"V on Arx, L","venue":null,"work_id":"a4c58bb4-b18f-4aa2-90f8-7f5860145055","year":2025},"citing_paper":{"arxiv_id":"2604.28139","last_updated":"2026-05-01T09:39:37Z","snapshot_observed_at":"2026-08-11T17:16:41.573956Z","submitted_at":"2026-04-30T17:23:19Z","title":"Claw-Eval-Live: A Live Agent Benchmark for Evolving Real-World Workflows","version":2},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-05-07T05:52:13.100867Z"},"links":{"citing_paper":"/paper/2604.28139"},"observation_digest":"sha256:f18ac3ed49ae4055f582f2c582eabbc14ba7c891b1d4bd413398bca008e8334b","observation_id":"4a64bbd9-42a5-44ec-bf17-ee101a015911","resolution":{"observed_at":"2026-05-27T10:23:57.841546Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.16741","last_updated":"2025-04-18T18:14:31Z","snapshot_observed_at":"2026-08-02T14:58:44.167588Z","submitted_at":"2024-07-23T17:50:43Z","title":"OpenHands: An Open Platform for AI Software Developers as Generalist Agents","version":3},"cited_work":{"arxiv_id":"2407.16741","doi":"10.1145/3718958.3750537","metadata_source":"pith","pith_arxiv_id":"2407.16741","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"OpenHands: An Open Platform for AI Software Developers as Generalist Agents","venue":"cs.SE","work_id":"f1762ea0-e382-4f38-a28c-adc643789859","year":2024},"citing_paper":{"arxiv_id":"2604.28139","last_updated":"2026-05-01T09:39:37Z","snapshot_observed_at":"2026-08-11T17:16:41.573956Z","submitted_at":"2026-04-30T17:23:19Z","title":"Claw-Eval-Live: A Live Agent Benchmark for Evolving Real-World Workflows","version":2},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-05-07T05:52:13.100867Z"},"links":{"cited_paper":"/paper/2407.16741","citing_paper":"/paper/2604.28139"},"observation_digest":"sha256:dd9381b56f3c001f093eec5c27e4fb6b49c70c6a998e5dd4941f45a18e21f363","observation_id":"04567d56-e05e-4d5b-8b4f-becf4244eb7b","resolution":{"observed_at":"2026-05-12T10:31:29.104724Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"adda25aa-2283-4db1-a5ad-1a680270fb2c","year":2024},"citing_paper":{"arxiv_id":"2604.28139","last_updated":"2026-05-01T09:39:37Z","snapshot_observed_at":"2026-08-11T17:16:41.573956Z","submitted_at":"2026-04-30T17:23:19Z","title":"Claw-Eval-Live: A Live Agent Benchmark for Evolving Real-World Workflows","version":2},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-05-07T05:52:13.100867Z"},"links":{"citing_paper":"/paper/2604.28139"},"observation_digest":"sha256:ec13917749ae359134a032b0d094d8c41847b0f405273068e6d174b02eafc6b6","observation_id":"1c13c64f-a847-417d-9695-c681a269dcc4","resolution":{"observed_at":"2026-05-27T10:28:57.847419Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2508.20453","last_updated":"2025-08-28T05:58:57Z","snapshot_observed_at":"2026-08-05T15:10:31.464813Z","submitted_at":"2025-08-28T05:58:57Z","title":"MCP-Bench: Benchmarking Tool-Using LLM Agents with Complex Real-World Tasks via MCP Servers","version":1},"cited_work":{"arxiv_id":"2508.20453","doi":"10.48550/arxiv.2508.20453","metadata_source":"arxiv_reference","pith_arxiv_id":"2508.20453","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Wang et al","venue":"ArXiv.org","work_id":"09a5d0b4-4a61-4a3e-a239-f3d9368305d5","year":2025},"citing_paper":{"arxiv_id":"2604.28139","last_updated":"2026-05-01T09:39:37Z","snapshot_observed_at":"2026-08-11T17:16:41.573956Z","submitted_at":"2026-04-30T17:23:19Z","title":"Claw-Eval-Live: A Live Agent Benchmark for Evolving Real-World Workflows","version":2},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-05-07T05:52:13.100867Z"},"links":{"cited_paper":"/paper/2508.20453","citing_paper":"/paper/2604.28139"},"observation_digest":"sha256:1a5285ff5733a2f065956e3994d855ed57dd951f38838aca03abd1343ef7a88b","observation_id":"8abc306b-5faa-49d8-82b3-e66dafa5c7c1","resolution":{"observed_at":"2026-05-12T10:31:29.155151Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"2faf1476-6de0-401f-933d-974fd17c4f32","year":2024},"citing_paper":{"arxiv_id":"2604.28139","last_updated":"2026-05-01T09:39:37Z","snapshot_observed_at":"2026-08-11T17:16:41.573956Z","submitted_at":"2026-04-30T17:23:19Z","title":"Claw-Eval-Live: A Live Agent Benchmark for Evolving Real-World Workflows","version":2},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-05-07T05:52:13.100867Z"},"links":{"citing_paper":"/paper/2604.28139"},"observation_digest":"sha256:5649ca2ca563084ee4df282dcf017fe9e520f327026f14602fc539342a19947a","observation_id":"7e85edcb-f619-4e0f-aeed-edab1a824a30","resolution":{"observed_at":"2026-05-27T10:23:57.847242Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"dbb77ce6-976d-453d-855f-f3c4176016ed","year":2024},"citing_paper":{"arxiv_id":"2604.28139","last_updated":"2026-05-01T09:39:37Z","snapshot_observed_at":"2026-08-11T17:16:41.573956Z","submitted_at":"2026-04-30T17:23:19Z","title":"Claw-Eval-Live: A Live Agent Benchmark for Evolving Real-World Workflows","version":2},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-05-07T05:52:13.100867Z"},"links":{"citing_paper":"/paper/2604.28139"},"observation_digest":"sha256:6501b0ed651883cec30c95162b5b7c3bb4ffd60f863141e010019465072f2f39","observation_id":"8fc5ca03-5ca6-40b2-8194-b74634e38570","resolution":{"observed_at":"2026-05-27T10:23:57.822591Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Xiong, Y","venue":null,"work_id":"1bb34d73-cba8-42e1-9d79-5ca7ae0bed23","year":2024},"citing_paper":{"arxiv_id":"2604.28139","last_updated":"2026-05-01T09:39:37Z","snapshot_observed_at":"2026-08-11T17:16:41.573956Z","submitted_at":"2026-04-30T17:23:19Z","title":"Claw-Eval-Live: A Live Agent Benchmark for Evolving Real-World Workflows","version":2},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-05-07T05:52:13.100867Z"},"links":{"citing_paper":"/paper/2604.28139"},"observation_digest":"sha256:4178990ee26030c4ed7281f66d6df7253100d6731091c57568f6d2826d2b9241","observation_id":"344709bb-892f-41fc-96bb-349febb2fa49","resolution":{"observed_at":"2026-05-27T10:23:57.825859Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"51bfaaee-b89f-494a-890d-d05e2ba71314","year":2025},"citing_paper":{"arxiv_id":"2604.28139","last_updated":"2026-05-01T09:39:37Z","snapshot_observed_at":"2026-08-11T17:16:41.573956Z","submitted_at":"2026-04-30T17:23:19Z","title":"Claw-Eval-Live: A Live Agent Benchmark for Evolving Real-World Workflows","version":2},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-05-07T05:52:13.100867Z"},"links":{"citing_paper":"/paper/2604.28139"},"observation_digest":"sha256:1dfed6856b0246650d3f8e7833c9142298d5a0a00145932f05425b62c036b871","observation_id":"3a905574-c8c0-41e5-9bf8-b6b6fe765eb9","resolution":{"observed_at":"2026-05-27T10:23:57.829545Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"4b06ef91-238a-4afc-b02a-79f0c065e7b5","year":2025},"citing_paper":{"arxiv_id":"2604.28139","last_updated":"2026-05-01T09:39:37Z","snapshot_observed_at":"2026-08-11T17:16:41.573956Z","submitted_at":"2026-04-30T17:23:19Z","title":"Claw-Eval-Live: A Live Agent Benchmark for Evolving Real-World Workflows","version":2},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-05-07T05:52:13.100867Z"},"links":{"citing_paper":"/paper/2604.28139"},"observation_digest":"sha256:05e7018693940d2cb3e028418faae54940f59d1088ea22e0a1d3a9b84a751b01","observation_id":"fb193e32-3090-4d29-b92d-626f4e3e501e","resolution":{"observed_at":"2026-05-27T10:23:57.832556Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"c711e698-add7-4534-8431-0e6f967f1be2","year":2023},"citing_paper":{"arxiv_id":"2604.28139","last_updated":"2026-05-01T09:39:37Z","snapshot_observed_at":"2026-08-11T17:16:41.573956Z","submitted_at":"2026-04-30T17:23:19Z","title":"Claw-Eval-Live: A Live Agent Benchmark for Evolving Real-World Workflows","version":2},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-05-07T05:52:13.100867Z"},"links":{"citing_paper":"/paper/2604.28139"},"observation_digest":"sha256:ea8e068e8885b4b86e2c8c0d1c49040d80a5aeaed37010664aa2d3c469454536","observation_id":"b3e27d8a-28e7-46df-8c0d-dd58b2def19b","resolution":{"observed_at":"2026-05-27T10:23:57.850452Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.12045","last_updated":"2024-06-17T19:33:08Z","snapshot_observed_at":"2026-08-08T21:08:39.676079Z","submitted_at":"2024-06-17T19:33:08Z","title":"$\\tau$-bench: A Benchmark for Tool-Agent-User Interaction in Real-World Domains","version":1},"cited_work":{"arxiv_id":"2406.12045","doi":"10.48550/arxiv.2406.12045","metadata_source":"pith","pith_arxiv_id":"2406.12045","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"$\\tau$-bench: A Benchmark for Tool-Agent-User Interaction in Real-World Domains","venue":"cs.AI","work_id":"6a8d8dc4-0cc0-4052-8109-abbcdcd4a962","year":2024},"citing_paper":{"arxiv_id":"2604.28139","last_updated":"2026-05-01T09:39:37Z","snapshot_observed_at":"2026-08-11T17:16:41.573956Z","submitted_at":"2026-04-30T17:23:19Z","title":"Claw-Eval-Live: A Live Agent Benchmark for Evolving Real-World Workflows","version":2},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-05-07T05:52:13.100867Z"},"links":{"cited_paper":"/paper/2406.12045","citing_paper":"/paper/2604.28139"},"observation_digest":"sha256:a6efa0e78610f563281c6cd30aa6fa2e4b0d20fccb5e5f4dc66f0a3697866127","observation_id":"487bbcac-ea85-41c0-a1d5-b67ca897f59e","resolution":{"observed_at":"2026-05-12T10:31:29.095205Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-07-14T18:20:21.86453+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-14T18:20:21.86453+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"3b40638e-3717-47b7-add4-fd6801ee4b38","year":2026},"citing_paper":{"arxiv_id":"2604.28139","last_updated":"2026-05-01T09:39:37Z","snapshot_observed_at":"2026-08-11T17:16:41.573956Z","submitted_at":"2026-04-30T17:23:19Z","title":"Claw-Eval-Live: A Live Agent Benchmark for Evolving Real-World Workflows","version":2},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-05-07T05:52:13.100867Z"},"links":{"citing_paper":"/paper/2604.28139"},"observation_digest":"sha256:77115b269d57e62c96e586a846f491aee33666e6a2e4bee392845c67751ba3d6","observation_id":"c026b1ae-1f5b-4983-bfa6-564d44cf5c4f","resolution":{"observed_at":"2026-05-27T10:23:57.856560Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Yoran, S","venue":null,"work_id":"f51048bf-6f0e-48fe-8071-4d56ac19b7bf","year":2024},"citing_paper":{"arxiv_id":"2604.28139","last_updated":"2026-05-01T09:39:37Z","snapshot_observed_at":"2026-08-11T17:16:41.573956Z","submitted_at":"2026-04-30T17:23:19Z","title":"Claw-Eval-Live: A Live Agent Benchmark for Evolving Real-World Workflows","version":2},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-05-07T05:52:13.100867Z"},"links":{"citing_paper":"/paper/2604.28139"},"observation_digest":"sha256:4910f932aca973bf596e96148ecc982a58014b5fee88e2991800996fcf0de3e6","observation_id":"55c8c62f-175f-4e8c-8bd4-ca042a21a757","resolution":{"observed_at":"2026-05-27T10:28:57.860614Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"2e5207b2-5332-402a-9a97-db74d2e5a385","year":2024},"citing_paper":{"arxiv_id":"2604.28139","last_updated":"2026-05-01T09:39:37Z","snapshot_observed_at":"2026-08-11T17:16:41.573956Z","submitted_at":"2026-04-30T17:23:19Z","title":"Claw-Eval-Live: A Live Agent Benchmark for Evolving Real-World Workflows","version":2},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-05-07T05:52:13.100867Z"},"links":{"citing_paper":"/paper/2604.28139"},"observation_digest":"sha256:4959d54c7006f24261c90188f9291fac9e7848a8df60dbe85ea07ec432e43edf","observation_id":"463352f7-d5fb-4dab-91d4-299179e13827","resolution":{"observed_at":"2026-05-27T10:23:57.819287Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Zhang, Y","venue":null,"work_id":"3b4b99ff-d843-4506-a759-ac5e80a17fab","year":2026},"citing_paper":{"arxiv_id":"2604.28139","last_updated":"2026-05-01T09:39:37Z","snapshot_observed_at":"2026-08-11T17:16:41.573956Z","submitted_at":"2026-04-30T17:23:19Z","title":"Claw-Eval-Live: A Live Agent Benchmark for Evolving Real-World Workflows","version":2},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-05-07T05:52:13.100867Z"},"links":{"citing_paper":"/paper/2604.28139"},"observation_digest":"sha256:8c76d88fe347e879ad171bc635c690829ea7564ad72840c46d77caa6dc20ed71","observation_id":"5efd8bb5-7755-4775-a66c-9cff94d41af2","resolution":{"observed_at":"2026-05-27T10:23:57.816574Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.14470","last_updated":"2025-05-20T05:58:23Z","snapshot_observed_at":"2026-08-06T12:35:19.109481Z","submitted_at":"2024-12-19T02:35:15Z","title":"Agent-SafetyBench: Evaluating the Safety of LLM Agents","version":2},"cited_work":{"arxiv_id":"2412.14470","doi":"10.48550/arxiv.2412.14470","metadata_source":"pith","pith_arxiv_id":"2412.14470","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Agent-SafetyBench: Evaluating the Safety of LLM Agents","venue":"cs.CL","work_id":"96afb8b9-0e7e-442c-93b1-6638599fc041","year":2024},"citing_paper":{"arxiv_id":"2604.28139","last_updated":"2026-05-01T09:39:37Z","snapshot_observed_at":"2026-08-11T17:16:41.573956Z","submitted_at":"2026-04-30T17:23:19Z","title":"Claw-Eval-Live: A Live Agent Benchmark for Evolving Real-World Workflows","version":2},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-05-07T05:52:13.100867Z"},"links":{"cited_paper":"/paper/2412.14470","citing_paper":"/paper/2604.28139"},"observation_digest":"sha256:4a8e21c0c4c154e407303cc38c2b904287437aa90ed0e4def4a1b4b9e2307b44","observation_id":"a5619918-88b9-4c78-b012-eeb1a4e24710","resolution":{"observed_at":"2026-05-13T06:30:47.228681Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-07-14T18:20:24.185748+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-14T18:20:24.185748+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.13854","last_updated":"2024-04-16T15:13:18Z","snapshot_observed_at":"2026-08-06T12:47:01.809185Z","submitted_at":"2023-07-25T22:59:32Z","title":"WebArena: A Realistic Web Environment for Building Autonomous Agents","version":4},"cited_work":{"arxiv_id":"2307.13854","doi":"10.48550/arxiv.2307.13854","metadata_source":"pith","pith_arxiv_id":"2307.13854","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"WebArena: A Realistic Web Environment for Building Autonomous Agents","venue":"cs.AI","work_id":"7058ffd2-a339-4102-89eb-248eeb074652","year":2023},"citing_paper":{"arxiv_id":"2604.28139","last_updated":"2026-05-01T09:39:37Z","snapshot_observed_at":"2026-08-11T17:16:41.573956Z","submitted_at":"2026-04-30T17:23:19Z","title":"Claw-Eval-Live: A Live Agent Benchmark for Evolving Real-World Workflows","version":2},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-05-07T05:52:13.100867Z"},"links":{"cited_paper":"/paper/2307.13854","citing_paper":"/paper/2604.28139"},"observation_digest":"sha256:9e7e0067c0b726e189c4304389c940d985b9efc98c07a759505710e3158878b1","observation_id":"4f704f51-0a2b-4875-9545-118aa23340b7","resolution":{"observed_at":"2026-05-12T10:31:29.139250Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-13T06:32:02.005865+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-05-20T18:52:18.85917+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-20T18:52:18.85917+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2604.28139","last_updated":"2026-05-01T09:39:37Z","latest_version":2,"primary_category":"cs.SE","snapshot_observed_at":"2026-08-11T17:16:41.573956Z","submitted_at":"2026-04-30T17:23:19Z","title":"Claw-Eval-Live: A Live Agent Benchmark for Evolving Real-World Workflows"},"reference_resolution":{"displayed":54,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":16,"verified_exact":24,"verified_fuzzy":14},"total_outbound_references":54},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-13T06:32:02.005865+00:00","source":"crossref"},{"observed_at":"2026-08-13T06:31:53.387327+00:00","source":"retraction_watch"}],"thesis":"As of 13 August 2026, this Paper Citation Record lists 54 of 54 outbound references and 9 inbound Pith citation observations for arXiv:2604.28139."}