{"as_of":"2026-08-07T08:17:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:47734d31bd1158672b160419741efe54ad55335271c335ff6ff57f61f142a11f","coverage":[{"denominator":23,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":23,"source":"paper_references, paper_reference_links","source_observed_at":"2026-05-10T16:43:27.037355Z","state":"measured"},{"denominator":26,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":26,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-07T06:34:17.273281+00:00","state":"measured"},{"denominator":3,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":3,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T00:13:59.086704Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"pith","source_observed_at":"2026-07-04T17:50:00.070007Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2604.10866","last_updated":"2026-04-16T16:00:33Z","snapshot_observed_at":"2026-07-06T22:59:23.126963Z","submitted_at":"2026-04-13T00:27:32Z","title":"OccuBench: Evaluating AI Agents on Real-World Professional Tasks via Language Environment Simulation","version":2},"cited_work":{"arxiv_id":"2604.10866","doi":null,"metadata_source":"pith","pith_arxiv_id":"2604.10866","snapshot_observed_at":"2026-07-04T17:50:00.070007Z","title":"OccuBench: Evaluating AI Agents on Real-World Professional Tasks via Language Environment Simulation","venue":"cs.CL","work_id":"713fa134-9489-4350-b25e-227ca53f1eba","year":2026},"citing_paper":{"arxiv_id":"2606.24669","last_updated":"2026-06-23T15:03:31Z","snapshot_observed_at":"2026-07-06T23:59:13.320720Z","submitted_at":"2026-06-23T15:03:31Z","title":"LaGO: Latent Action Guidance for Online Reinforcement Learning","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-06-25T23:26:55.307583Z"},"links":{"cited_paper":"/paper/2604.10866","citing_paper":"/paper/2606.24669"},"observation_digest":"sha256:27440254eff5a950c64839df54f473d51df50ed4a08741afb5f26ba8b883d859","observation_id":"1ef7c2d8-170f-4967-9f36-10c905273466","resolution":{"observed_at":"2026-07-04T17:50:00.071339Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2604.10866","last_updated":"2026-04-16T16:00:33Z","snapshot_observed_at":"2026-07-06T22:59:23.126963Z","submitted_at":"2026-04-13T00:27:32Z","title":"OccuBench: Evaluating AI Agents on Real-World Professional Tasks via Language Environment Simulation","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2604.10866","snapshot_observed_at":"2026-08-02T14:50:11.936926Z","title":"Occubench: Evaluating ai agents on real-world professional tasks via language world models.arXiv preprint arXiv:2604.10866, 2026","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2607.16204","last_updated":"2026-05-07T00:40:32Z","snapshot_observed_at":"2026-08-06T07:49:27.049032Z","submitted_at":"2026-05-07T00:40:32Z","title":"Masked Diffusion Language Models are Strong and Steerable Text-Based World Models for Agentic RL","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-02T14:50:11.936926Z"},"links":{"cited_paper":"/paper/2604.10866","citing_paper":"/paper/2607.16204"},"observation_digest":"sha256:edc6120420696627b1db97f09e11208228bc7b80bf3eb32cb3ea9ea08766b734","observation_id":"93ed0ce3-6533-48a8-851a-9aa3c8f22b85","resolution":{"observed_at":"2026-08-02T14:50:11.936926Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2604.10866","last_updated":"2026-04-16T16:00:33Z","snapshot_observed_at":"2026-07-06T22:59:23.126963Z","submitted_at":"2026-04-13T00:27:32Z","title":"OccuBench: Evaluating AI Agents on Real-World Professional Tasks via Language Environment Simulation","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2604.10866","snapshot_observed_at":"2026-08-07T00:13:59.086704Z","title":"OccuBench: Evaluating AI agents on real-world professional tasks via language world models.arXiv preprint arXiv:2604.10866, 2026","venue":null,"work_id":null,"year":2026},"citing_paper":{"arxiv_id":"2608.02713","last_updated":"2026-08-03T17:59:58Z","snapshot_observed_at":"2026-08-07T08:13:26.569151Z","submitted_at":"2026-08-03T17:59:58Z","title":"Quo Vadis, World Modeling?","version":1},"reference_index":70,"source":"pdf_text","source_observed_at":"2026-08-07T00:13:59.086704Z"},"links":{"cited_paper":"/paper/2604.10866","citing_paper":"/paper/2608.02713"},"observation_digest":"sha256:0f7ed8915c4e65cc16110091e942fdc4c21fb71266fae9d1d2688fd7fd89c6a7","observation_id":"ad3f85ea-6b51-4094-a9a1-7b753821c9f5","resolution":{"observed_at":"2026-08-07T00:13:59.086704Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2604.10866/citation-record","integrity":"/paper/2604.10866/integrity","json":"/paper/2604.10866/citation-record.json","paper":"/paper/2604.10866"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2602.00933","last_updated":"2026-05-19T23:26:22Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2026-01-31T23:19:39Z","title":"MCP-Atlas: A Large-Scale Benchmark for Tool-Use Competency with Real MCP Servers","version":3},"cited_work":{"arxiv_id":"2602.00933","doi":null,"metadata_source":"pith","pith_arxiv_id":"2602.00933","snapshot_observed_at":"2026-07-04T07:39:39.044133Z","title":"MCP-Atlas: A Large-Scale Benchmark for Tool-Use Competency with Real MCP Servers","venue":"cs.SE","work_id":"1faed39f-9227-4673-94df-80af7e9aa303","year":2026},"citing_paper":{"arxiv_id":"2604.10866","last_updated":"2026-04-16T16:00:33Z","snapshot_observed_at":"2026-07-06T22:59:23.126963Z","submitted_at":"2026-04-13T00:27:32Z","title":"OccuBench: Evaluating AI Agents on Real-World Professional Tasks via Language Environment Simulation","version":2},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-10T16:43:27.037355Z"},"links":{"cited_paper":"/paper/2602.00933","citing_paper":"/paper/2604.10866"},"observation_digest":"sha256:9f1da43aa308e17e0ca811db64dd6b3fdcbfaecf3c6b45d25f20703866acb5c9","observation_id":"5c445c85-7ae2-432b-b043-7b93b33415ec","resolution":{"observed_at":"2026-05-11T08:20:57.821340Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2510.15047","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-04T19:40:06.149325Z","title":"Internalizing world models via self-play finetuning for agentic RL.arXiv preprint arXiv:2510.15047","venue":null,"work_id":"f4defb5e-8e63-4dae-91bc-8581704f2ab3","year":2025},"citing_paper":{"arxiv_id":"2604.10866","last_updated":"2026-04-16T16:00:33Z","snapshot_observed_at":"2026-07-06T22:59:23.126963Z","submitted_at":"2026-04-13T00:27:32Z","title":"OccuBench: Evaluating AI Agents on Real-World Professional Tasks via Language Environment Simulation","version":2},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-10T16:43:27.037355Z"},"links":{"citing_paper":"/paper/2604.10866"},"observation_digest":"sha256:dd89b67ad346666d5bf20a8dff1a698e0b73342ca4043c3940b9e6f1076e7c95","observation_id":"802a4a1f-9dd6-4778-8cf7-4247fae0ace8","resolution":{"observed_at":"2026-05-11T08:20:57.915436Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2512.02556","last_updated":"2025-12-02T09:25:14Z","snapshot_observed_at":"2026-07-31T23:49:25.878472Z","submitted_at":"2025-12-02T09:25:14Z","title":"DeepSeek-V3.2: Pushing the Frontier of Open Large Language Models","version":1},"cited_work":{"arxiv_id":"2512.02556","doi":"10.18653/v1/d18-1512","metadata_source":"pith","pith_arxiv_id":"2512.02556","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"DeepSeek-V3.2: Pushing the Frontier of Open Large Language Models","venue":"cs.CL","work_id":"07c85cc5-4086-4abc-823b-6d0f4ff784d0","year":2025},"citing_paper":{"arxiv_id":"2604.10866","last_updated":"2026-04-16T16:00:33Z","snapshot_observed_at":"2026-07-06T22:59:23.126963Z","submitted_at":"2026-04-13T00:27:32Z","title":"OccuBench: Evaluating AI Agents on Real-World Professional Tasks via Language Environment Simulation","version":2},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-10T16:43:27.037355Z"},"links":{"cited_paper":"/paper/2512.02556","citing_paper":"/paper/2604.10866"},"observation_digest":"sha256:c2f38c140aa508c805db288c0a749cf416ea1124aef18b15f29a949fd8312a5b","observation_id":"c44ffc03-2ed3-4aea-bab0-c0f43cefca63","resolution":{"observed_at":"2026-05-11T08:20:57.872101Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.00993","last_updated":"2024-07-01T06:10:01Z","snapshot_observed_at":"2026-07-06T18:39:20.664293Z","submitted_at":"2024-07-01T06:10:01Z","title":"Mobile-Bench: An Evaluation Benchmark for LLM-based Mobile Agents","version":1},"cited_work":{"arxiv_id":"2407.00993","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2407.00993","snapshot_observed_at":"2026-06-30T07:14:21.229715Z","title":"Mobile-Bench: An Evaluation Benchmark for LLM-based Mobile Agents, July 2024","venue":null,"work_id":"f732e035-6f78-4772-bb52-681e9533b1d1","year":2024},"citing_paper":{"arxiv_id":"2604.10866","last_updated":"2026-04-16T16:00:33Z","snapshot_observed_at":"2026-07-06T22:59:23.126963Z","submitted_at":"2026-04-13T00:27:32Z","title":"OccuBench: Evaluating AI Agents on Real-World Professional Tasks via Language Environment Simulation","version":2},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-10T16:43:27.037355Z"},"links":{"cited_paper":"/paper/2407.00993","citing_paper":"/paper/2604.10866"},"observation_digest":"sha256:43f3fffefb7bbc5ed029461c67d779f23e00d69b68cb3837f3530d5f479dcd44","observation_id":"94a5adaf-1242-4b0c-af22-d1c020b95e45","resolution":{"observed_at":"2026-05-11T08:20:57.851105Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2602.03587","doi":"10.48550/arxiv.2602.03587","metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Cl-bench: A benchmark for context learning","venue":"Open MIND","work_id":"4d10e4c6-07c6-47a2-bf06-94046a4b3d1a","year":2026},"citing_paper":{"arxiv_id":"2604.10866","last_updated":"2026-04-16T16:00:33Z","snapshot_observed_at":"2026-07-06T22:59:23.126963Z","submitted_at":"2026-04-13T00:27:32Z","title":"OccuBench: Evaluating AI Agents on Real-World Professional Tasks via Language Environment Simulation","version":2},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-05-10T16:43:27.037355Z"},"links":{"citing_paper":"/paper/2604.10866"},"observation_digest":"sha256:77eac9782a7f3e290dcc961cc069217f8eadc33990c3fbf4e1a42772bcd49f4c","observation_id":"2f90b6c4-3ccf-42b9-87d7-f15b37d8cc2b","resolution":{"observed_at":"2026-05-11T08:20:57.920830Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2602.15763","last_updated":"2026-02-24T10:44:44Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2026-02-17T17:50:56Z","title":"GLM-5: from Vibe Coding to Agentic Engineering","version":2},"cited_work":{"arxiv_id":"2602.15763","doi":"10.48550/arxiv.2602.15763","metadata_source":"pith","pith_arxiv_id":"2602.15763","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"GLM-5: from Vibe Coding to Agentic Engineering","venue":"cs.LG","work_id":"ad29b1a2-bf77-46b3-9ead-fb62b1d2c6fe","year":2026},"citing_paper":{"arxiv_id":"2604.10866","last_updated":"2026-04-16T16:00:33Z","snapshot_observed_at":"2026-07-06T22:59:23.126963Z","submitted_at":"2026-04-13T00:27:32Z","title":"OccuBench: Evaluating AI Agents on Real-World Professional Tasks via Language Environment Simulation","version":2},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-05-10T16:43:27.037355Z"},"links":{"cited_paper":"/paper/2602.15763","citing_paper":"/paper/2604.10866"},"observation_digest":"sha256:1a3bfc58d3eec40a35142d740a07e23000172f14a796b230223d15301ec29520","observation_id":"af91293d-5720-4698-8270-78209db53825","resolution":{"observed_at":"2026-05-11T08:20:57.903585Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.06559","last_updated":"2025-04-01T05:04:47Z","snapshot_observed_at":"2026-07-06T19:48:10.800324Z","submitted_at":"2024-11-10T18:50:51Z","title":"Is Your LLM Secretly a World Model of the Internet? Model-Based Planning for Web Agents","version":2},"cited_work":{"arxiv_id":"2411.06559","doi":"10.48550/arxiv.2411.06559","metadata_source":"arxiv_reference","pith_arxiv_id":"2411.06559","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Is your llm secretly a world model of the internet? model-based planning for web agents","venue":"arXiv (Cornell University)","work_id":"30c2e3ae-88b7-41e6-986b-6f097fe53d5f","year":2024},"citing_paper":{"arxiv_id":"2604.10866","last_updated":"2026-04-16T16:00:33Z","snapshot_observed_at":"2026-07-06T22:59:23.126963Z","submitted_at":"2026-04-13T00:27:32Z","title":"OccuBench: Evaluating AI Agents on Real-World Professional Tasks via Language Environment Simulation","version":2},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-05-10T16:43:27.037355Z"},"links":{"cited_paper":"/paper/2411.06559","citing_paper":"/paper/2604.10866"},"observation_digest":"sha256:05cfb6654e34e0ca536e6fcaeee52a16132beb0d52b326badc4600247d2456d8","observation_id":"fc736636-c5c8-4df0-9680-a9466798b0bc","resolution":{"observed_at":"2026-05-11T08:20:57.887889Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2602.02276","last_updated":"2026-02-02T16:17:38Z","snapshot_observed_at":"2026-07-06T22:44:09.804048Z","submitted_at":"2026-02-02T16:17:38Z","title":"Kimi K2.5: Visual Agentic Intelligence","version":1},"cited_work":{"arxiv_id":"2602.02276","doi":"10.48550/arxiv.2602.02276","metadata_source":"pith","pith_arxiv_id":"2602.02276","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Kimi K2.5: Visual Agentic Intelligence","venue":"cs.CL","work_id":"d690be8f-5d53-49b0-b1e7-79668eb8fcdb","year":2026},"citing_paper":{"arxiv_id":"2604.10866","last_updated":"2026-04-16T16:00:33Z","snapshot_observed_at":"2026-07-06T22:59:23.126963Z","submitted_at":"2026-04-13T00:27:32Z","title":"OccuBench: Evaluating AI Agents on Real-World Professional Tasks via Language Environment Simulation","version":2},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-05-10T16:43:27.037355Z"},"links":{"cited_paper":"/paper/2602.02276","citing_paper":"/paper/2604.10866"},"observation_digest":"sha256:1e23f7c63b9ed1c8b325d2a492d9de5185a3806183b1f32c9b3aff1959b73da8","observation_id":"5c9ef5b4-240d-48e6-b590-7ff9a4f97561","resolution":{"observed_at":"2026-05-11T08:20:57.834346Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2511.01824","doi":"10.48550/arxiv.2511.01824","metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Simulating environments with reasoning models for agent training","venue":"ArXiv.org","work_id":"61b420ed-1da3-4248-86be-712adc54bba2","year":2025},"citing_paper":{"arxiv_id":"2604.10866","last_updated":"2026-04-16T16:00:33Z","snapshot_observed_at":"2026-07-06T22:59:23.126963Z","submitted_at":"2026-04-13T00:27:32Z","title":"OccuBench: Evaluating AI Agents on Real-World Professional Tasks via Language Environment Simulation","version":2},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-05-10T16:43:27.037355Z"},"links":{"citing_paper":"/paper/2604.10866"},"observation_digest":"sha256:f212ad0c1faaf7faeaa7df0428315f19513d1fd7a6db5b4b5a938c8ea4cf8fca","observation_id":"070a4869-ea72-4541-a805-2f59d9baadab","resolution":{"observed_at":"2026-05-11T08:20:57.860037Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.13936","last_updated":"2025-05-20T14:11:37Z","snapshot_observed_at":"2026-07-06T21:11:39.503336Z","submitted_at":"2025-04-15T14:03:10Z","title":"ViMo: A Generative Visual GUI World Model for App Agents","version":2},"cited_work":{"arxiv_id":"2504.13936","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2504.13936","snapshot_observed_at":"2026-07-04T17:29:59.601323Z","title":"Vimo: A generative visual gui world model for app agents","venue":null,"work_id":"06633d92-fd4d-41aa-a914-35ffc813c3a8","year":2025},"citing_paper":{"arxiv_id":"2604.10866","last_updated":"2026-04-16T16:00:33Z","snapshot_observed_at":"2026-07-06T22:59:23.126963Z","submitted_at":"2026-04-13T00:27:32Z","title":"OccuBench: Evaluating AI Agents on Real-World Professional Tasks via Language Environment Simulation","version":2},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-05-10T16:43:27.037355Z"},"links":{"cited_paper":"/paper/2504.13936","citing_paper":"/paper/2604.10866"},"observation_digest":"sha256:4b65c918e2cf26e989811376df8045c2047518cc8e7de8d418b2046246858408","observation_id":"9ef63fb1-351e-476c-b9e9-1c0ad813fb96","resolution":{"observed_at":"2026-05-11T08:20:57.898393Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2601.11868","last_updated":"2026-01-17T01:29:30Z","snapshot_observed_at":"2026-07-06T22:41:58.373427Z","submitted_at":"2026-01-17T01:29:30Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","version":1},"cited_work":{"arxiv_id":"2601.11868","doi":"10.48550/arxiv.2302.01973","metadata_source":"pith","pith_arxiv_id":"2601.11868","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","venue":"cs.SE","work_id":"0624be05-1d97-4fd6-8300-b04b8a3ab04b","year":2026},"citing_paper":{"arxiv_id":"2604.10866","last_updated":"2026-04-16T16:00:33Z","snapshot_observed_at":"2026-07-06T22:59:23.126963Z","submitted_at":"2026-04-13T00:27:32Z","title":"OccuBench: Evaluating AI Agents on Real-World Professional Tasks via Language Environment Simulation","version":2},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-05-10T16:43:27.037355Z"},"links":{"cited_paper":"/paper/2601.11868","citing_paper":"/paper/2604.10866"},"observation_digest":"sha256:e37693c15d5313ce8e5db92d74a2d6069313e9faffdddf58c49c0b52a0ed8309","observation_id":"d1cb947c-175e-4db4-9988-9380e13e9586","resolution":{"observed_at":"2026-05-11T08:20:57.808333Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.12983","last_updated":"2023-11-21T20:34:47Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-11-21T20:34:47Z","title":"GAIA: a benchmark for General AI Assistants","version":1},"cited_work":{"arxiv_id":"2311.12983","doi":"10.48550/arxiv.2311.12983","metadata_source":"pith","pith_arxiv_id":"2311.12983","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"GAIA: a benchmark for General AI Assistants","venue":"cs.CL","work_id":"cf222b33-f7a3-4044-a570-ecfe25edb3f8","year":2023},"citing_paper":{"arxiv_id":"2604.10866","last_updated":"2026-04-16T16:00:33Z","snapshot_observed_at":"2026-07-06T22:59:23.126963Z","submitted_at":"2026-04-13T00:27:32Z","title":"OccuBench: Evaluating AI Agents on Real-World Professional Tasks via Language Environment Simulation","version":2},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-05-10T16:43:27.037355Z"},"links":{"cited_paper":"/paper/2311.12983","citing_paper":"/paper/2604.10866"},"observation_digest":"sha256:47f99c2f11f643ea17d483de74151f8b785e02680ada6f46f31c34f7416cd85a","observation_id":"e1d17204-c528-4417-86f1-e567da618b05","resolution":{"observed_at":"2026-05-12T15:46:03.881709Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-05-20T18:52:17.222295+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-20T18:52:17.222295+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.12115","last_updated":"2025-05-29T23:07:34Z","snapshot_observed_at":"2026-07-06T20:38:00.177933Z","submitted_at":"2025-02-17T18:41:16Z","title":"SWE-Lancer: Can Frontier LLMs Earn $1 Million from Real-World Freelance Software Engineering?","version":4},"cited_work":{"arxiv_id":"2502.12115","doi":"10.48550/arxiv.2502.12115","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.12115","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SWE-Lancer: Can frontier LLMs earn $1 million from real-world freelance soft- ware engineering?","venue":"ArXiv.org","work_id":"160777a0-b67f-4dac-ab53-0b8d9d1cad82","year":2025},"citing_paper":{"arxiv_id":"2604.10866","last_updated":"2026-04-16T16:00:33Z","snapshot_observed_at":"2026-07-06T22:59:23.126963Z","submitted_at":"2026-04-13T00:27:32Z","title":"OccuBench: Evaluating AI Agents on Real-World Professional Tasks via Language Environment Simulation","version":2},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-05-10T16:43:27.037355Z"},"links":{"cited_paper":"/paper/2502.12115","citing_paper":"/paper/2604.10866"},"observation_digest":"sha256:ed41d62467a5ebc0848debe5091c8ea6a9e7fc55f2a99586f051fe2128fd2c88","observation_id":"f6055ae1-9489-40bc-b083-a8db4e8816d9","resolution":{"observed_at":"2026-05-11T08:20:58.016292Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2601.03267","last_updated":"2026-05-01T23:55:43Z","snapshot_observed_at":"2026-08-02T10:52:10.211700Z","submitted_at":"2025-12-19T07:05:38Z","title":"OpenAI GPT-5 System Card","version":2},"cited_work":{"arxiv_id":"2601.03267","doi":"10.48550/arxiv.2601.03267","metadata_source":"pith","pith_arxiv_id":"2601.03267","snapshot_observed_at":"2026-08-05T02:49:54.815029Z","title":"OpenAI GPT-5 System Card","venue":"cs.CL","work_id":"ca87689a-0d29-4476-b504-b65dbbb08af4","year":2025},"citing_paper":{"arxiv_id":"2604.10866","last_updated":"2026-04-16T16:00:33Z","snapshot_observed_at":"2026-07-06T22:59:23.126963Z","submitted_at":"2026-04-13T00:27:32Z","title":"OccuBench: Evaluating AI Agents on Real-World Professional Tasks via Language Environment Simulation","version":2},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-05-10T16:43:27.037355Z"},"links":{"cited_paper":"/paper/2601.03267","citing_paper":"/paper/2604.10866"},"observation_digest":"sha256:fd93cacab9d5e37329a417e5ba2f8449ad77a19ea75c1dbeafd71620f3a8590e","observation_id":"2be0ed0e-7423-401e-9fc9-da6612ca8508","resolution":{"observed_at":"2026-05-11T08:20:57.984885Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-03T00:38:11.458508+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-03T00:38:11.458508+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.03442","last_updated":"2023-08-06T00:21:19Z","snapshot_observed_at":"2026-07-06T15:13:13.695535Z","submitted_at":"2023-04-07T01:55:19Z","title":"Generative Agents: Interactive Simulacra of Human Behavior","version":2},"cited_work":{"arxiv_id":"2304.03442","doi":"10.1001/jamapsychiatry.2022.0609","metadata_source":"pith","pith_arxiv_id":"2304.03442","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Generative Agents: Interactive Simulacra of Human Behavior","venue":"cs.HC","work_id":"01f7ddaa-284a-441a-be87-921aad4dc54b","year":2023},"citing_paper":{"arxiv_id":"2604.10866","last_updated":"2026-04-16T16:00:33Z","snapshot_observed_at":"2026-07-06T22:59:23.126963Z","submitted_at":"2026-04-13T00:27:32Z","title":"OccuBench: Evaluating AI Agents on Real-World Professional Tasks via Language Environment Simulation","version":2},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-05-10T16:43:27.037355Z"},"links":{"cited_paper":"/paper/2304.03442","citing_paper":"/paper/2604.10866"},"observation_digest":"sha256:63a7998ecf3b28328a92829216bec3e8bee45a05cca5330ac461dcb7de5e74f9","observation_id":"737c3d97-8bf7-4b8f-acd5-3aebad835fb5","resolution":{"observed_at":"2026-05-11T19:05:14.047699Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-05-24T05:54:33.103795+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-24T05:54:33.103795+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2510.04374","last_updated":"2025-10-05T21:36:43Z","snapshot_observed_at":"2026-07-06T22:31:44.964774Z","submitted_at":"2025-10-05T21:36:43Z","title":"GDPval: Evaluating AI Model Performance on Real-World Economically Valuable Tasks","version":1},"cited_work":{"arxiv_id":"2510.04374","doi":"10.3102/0013189x023002005","metadata_source":"pith","pith_arxiv_id":"2510.04374","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"GDPval: Evaluating AI Model Performance on Real-World Economically Valuable Tasks","venue":"cs.LG","work_id":"6eca346e-0e48-4240-961c-1ecee1e71aab","year":2025},"citing_paper":{"arxiv_id":"2604.10866","last_updated":"2026-04-16T16:00:33Z","snapshot_observed_at":"2026-07-06T22:59:23.126963Z","submitted_at":"2026-04-13T00:27:32Z","title":"OccuBench: Evaluating AI Agents on Real-World Professional Tasks via Language Environment Simulation","version":2},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-05-10T16:43:27.037355Z"},"links":{"cited_paper":"/paper/2510.04374","citing_paper":"/paper/2604.10866"},"observation_digest":"sha256:4a9310f24f878fdf4eaea68431766f93cdc73645a3292397da734b4439dad03f","observation_id":"0c34966c-72ce-4aeb-b1fa-b9388f0d54d1","resolution":{"observed_at":"2026-05-16T11:11:04.750341Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2508.20453","last_updated":"2025-08-28T05:58:57Z","snapshot_observed_at":"2026-08-05T15:10:31.464813Z","submitted_at":"2025-08-28T05:58:57Z","title":"MCP-Bench: Benchmarking Tool-Using LLM Agents with Complex Real-World Tasks via MCP Servers","version":1},"cited_work":{"arxiv_id":"2508.20453","doi":"10.48550/arxiv.2508.20453","metadata_source":"arxiv_reference","pith_arxiv_id":"2508.20453","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Wang et al","venue":"ArXiv.org","work_id":"09a5d0b4-4a61-4a3e-a239-f3d9368305d5","year":2025},"citing_paper":{"arxiv_id":"2604.10866","last_updated":"2026-04-16T16:00:33Z","snapshot_observed_at":"2026-07-06T22:59:23.126963Z","submitted_at":"2026-04-13T00:27:32Z","title":"OccuBench: Evaluating AI Agents on Real-World Professional Tasks via Language Environment Simulation","version":2},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-05-10T16:43:27.037355Z"},"links":{"cited_paper":"/paper/2508.20453","citing_paper":"/paper/2604.10866"},"observation_digest":"sha256:fad34a1a5f4fddf139e36f1d6b961eb84c5b6b11c9f88644b7c47646482e72dd","observation_id":"cfdf2bd0-6e26-4321-9851-8e8e7855c463","resolution":{"observed_at":"2026-05-11T08:20:58.062227Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.12516","last_updated":"2025-04-16T22:27:45Z","snapshot_observed_at":"2026-08-03T00:43:33.338074Z","submitted_at":"2025-04-16T22:27:45Z","title":"BrowseComp: A Simple Yet Challenging Benchmark for Browsing Agents","version":1},"cited_work":{"arxiv_id":"2504.12516","doi":"10.48550/arxiv.2504.12516","metadata_source":"pith","pith_arxiv_id":"2504.12516","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"BrowseComp: A Simple Yet Challenging Benchmark for Browsing Agents","venue":"cs.CL","work_id":"25adb508-d97c-49d6-ae43-7a70c2478a34","year":2025},"citing_paper":{"arxiv_id":"2604.10866","last_updated":"2026-04-16T16:00:33Z","snapshot_observed_at":"2026-07-06T22:59:23.126963Z","submitted_at":"2026-04-13T00:27:32Z","title":"OccuBench: Evaluating AI Agents on Real-World Professional Tasks via Language Environment Simulation","version":2},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-05-10T16:43:27.037355Z"},"links":{"cited_paper":"/paper/2504.12516","citing_paper":"/paper/2604.10866"},"observation_digest":"sha256:573e4ead613a51f32455e00e61ac676fba8c464559f74cae0bfbfc6c700916e0","observation_id":"813d494a-e6e8-4ce6-bc8f-18e8949d23f8","resolution":{"observed_at":"2026-05-12T07:44:32.193354Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2509.24002","doi":"10.48550/arxiv.2509.24002","metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mcpmark: A benchmark for stress-testing realistic and comprehensive mcp use","venue":"ArXiv.org","work_id":"5b1fa0f7-d604-4d0f-85f2-a2e91356834d","year":2025},"citing_paper":{"arxiv_id":"2604.10866","last_updated":"2026-04-16T16:00:33Z","snapshot_observed_at":"2026-07-06T22:59:23.126963Z","submitted_at":"2026-04-13T00:27:32Z","title":"OccuBench: Evaluating AI Agents on Real-World Professional Tasks via Language Environment Simulation","version":2},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-05-10T16:43:27.037355Z"},"links":{"citing_paper":"/paper/2604.10866"},"observation_digest":"sha256:0c6f214f0730d0263a2af4b894009cce398e64a784bca803f8a9a078a3c84b40","observation_id":"742b2b78-558f-453f-98c2-83c6bc9d7a6d","resolution":{"observed_at":"2026-05-11T08:20:58.111675Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2602.14721","doi":"10.48550/arxiv.2602.14721","metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Webworld: A large-scale world model for web agent training","venue":"Open MIND","work_id":"9a41f84f-9a89-4bcf-8b7c-2d9631db5fb2","year":2026},"citing_paper":{"arxiv_id":"2604.10866","last_updated":"2026-04-16T16:00:33Z","snapshot_observed_at":"2026-07-06T22:59:23.126963Z","submitted_at":"2026-04-13T00:27:32Z","title":"OccuBench: Evaluating AI Agents on Real-World Professional Tasks via Language Environment Simulation","version":2},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-05-10T16:43:27.037355Z"},"links":{"citing_paper":"/paper/2604.10866"},"observation_digest":"sha256:8d50007168cdbe6eed74627ab1f14c6ca09427e46377a030293e26a0bee2ba92","observation_id":"a4bcc152-078d-48fe-9d69-598b237cfe14","resolution":{"observed_at":"2026-05-11T08:20:58.008293Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2603.07980","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-04T20:00:08.402369Z","title":"$OneMillion-Bench: How far are language agents from human experts?arXiv preprint arXiv:2603.07980","venue":null,"work_id":"8519ff23-797d-41f2-834f-3bd0f249949d","year":2026},"citing_paper":{"arxiv_id":"2604.10866","last_updated":"2026-04-16T16:00:33Z","snapshot_observed_at":"2026-07-06T22:59:23.126963Z","submitted_at":"2026-04-13T00:27:32Z","title":"OccuBench: Evaluating AI Agents on Real-World Professional Tasks via Language Environment Simulation","version":2},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-05-10T16:43:27.037355Z"},"links":{"citing_paper":"/paper/2604.10866"},"observation_digest":"sha256:d6f9cada37296b63b9b5244680e114db95ccbe4cb1bb4e89d141cf1828e91e5b","observation_id":"83543edd-fa4c-4576-998b-f8fcd193ece6","resolution":{"observed_at":"2026-05-11T08:20:58.126776Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.12045","last_updated":"2024-06-17T19:33:08Z","snapshot_observed_at":"2026-08-02T22:19:29.043854Z","submitted_at":"2024-06-17T19:33:08Z","title":"$\\tau$-bench: A Benchmark for Tool-Agent-User Interaction in Real-World Domains","version":1},"cited_work":{"arxiv_id":"2406.12045","doi":"10.48550/arxiv.2406.12045","metadata_source":"pith","pith_arxiv_id":"2406.12045","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"$\\tau$-bench: A Benchmark for Tool-Agent-User Interaction in Real-World Domains","venue":"cs.AI","work_id":"6a8d8dc4-0cc0-4052-8109-abbcdcd4a962","year":2024},"citing_paper":{"arxiv_id":"2604.10866","last_updated":"2026-04-16T16:00:33Z","snapshot_observed_at":"2026-07-06T22:59:23.126963Z","submitted_at":"2026-04-13T00:27:32Z","title":"OccuBench: Evaluating AI Agents on Real-World Professional Tasks via Language Environment Simulation","version":2},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-05-10T16:43:27.037355Z"},"links":{"cited_paper":"/paper/2406.12045","citing_paper":"/paper/2604.10866"},"observation_digest":"sha256:bd2f588cba1cbbbb3d4f0d02ee7b2ee315ec5e3f765eb476d73a0d5689c763b6","observation_id":"5aab5c21-e55e-4c8f-a900-81e7e61dd8ee","resolution":{"observed_at":"2026-05-11T08:20:58.037323Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-07-14T18:20:21.86453+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-14T18:20:21.86453+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2604.06132","last_updated":"2026-05-07T14:06:23Z","snapshot_observed_at":"2026-07-06T22:54:40.349479Z","submitted_at":"2026-04-07T17:43:18Z","title":"Claw-Eval: Towards Trustworthy Evaluation of Autonomous Agents","version":3},"cited_work":{"arxiv_id":"2604.06132","doi":"10.48550/arxiv.2604.06132","metadata_source":"pith","pith_arxiv_id":"2604.06132","snapshot_observed_at":"2026-08-05T02:49:54.815029Z","title":"Claw-Eval: Towards Trustworthy Evaluation of Autonomous Agents","venue":"cs.AI","work_id":"57acc3ec-f4c3-49ab-bd0f-5aab91002df9","year":2026},"citing_paper":{"arxiv_id":"2604.10866","last_updated":"2026-04-16T16:00:33Z","snapshot_observed_at":"2026-07-06T22:59:23.126963Z","submitted_at":"2026-04-13T00:27:32Z","title":"OccuBench: Evaluating AI Agents on Real-World Professional Tasks via Language Environment Simulation","version":2},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-05-10T16:43:27.037355Z"},"links":{"cited_paper":"/paper/2604.06132","citing_paper":"/paper/2604.10866"},"observation_digest":"sha256:f459ce0d9d869ed30bd0fe843df6e5e278fbddd0f83fe7a031f55d17a7350382","observation_id":"8bd3fc51-c1ea-4313-8db6-6a3b6e6948e7","resolution":{"observed_at":"2026-05-11T08:20:57.970625Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2604.10866","last_updated":"2026-04-16T16:00:33Z","latest_version":2,"primary_category":"cs.CL","snapshot_observed_at":"2026-07-06T22:59:23.126963Z","submitted_at":"2026-04-13T00:27:32Z","title":"OccuBench: Evaluating AI Agents on Real-World Professional Tasks via Language Environment Simulation"},"reference_resolution":{"displayed":23,"state_counts":{"malformed_identifier":0,"metadata_mismatch":1,"parse_uncertain":0,"unresolved":0,"verified_exact":22,"verified_fuzzy":0},"total_outbound_references":23},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"thesis":"As of 7 August 2026, this Paper Citation Record lists 23 of 23 outbound references and 3 inbound Pith citation observations for arXiv:2604.10866."}