{"as_of":"2026-08-06T19:30:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:8c60ad051e9116d3e76c9655d731ddf336754a949ff1fe72f27ef773dd6670e2","coverage":[{"denominator":21,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":21,"source":"paper_references, paper_reference_links","source_observed_at":"2026-05-10T16:30:51.886471Z","state":"measured"},{"denominator":22,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":22,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-06T06:34:29.942622+00:00","state":"measured"},{"denominator":1,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":1,"source":"paper_references, paper_reference_links","source_observed_at":"2026-06-28T06:09:38.698353Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"pith","source_observed_at":"2026-07-02T08:16:48.330790Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2604.12162","last_updated":"2026-04-14T00:43:20Z","snapshot_observed_at":"2026-07-31T20:58:59.645474Z","submitted_at":"2026-04-14T00:43:20Z","title":"AlphaEval: Evaluating Agents in Production","version":1},"cited_work":{"arxiv_id":"2604.12162","doi":null,"metadata_source":"pith","pith_arxiv_id":"2604.12162","snapshot_observed_at":"2026-07-02T08:16:48.330790Z","title":"AlphaEval: Evaluating Agents in Production","venue":"cs.CL","work_id":"265364d8-dc18-4292-ab9a-220afba5e37a","year":2026},"citing_paper":{"arxiv_id":"2606.05342","last_updated":"2026-06-05T17:19:56Z","snapshot_observed_at":"2026-07-06T23:45:23.749718Z","submitted_at":"2026-06-03T18:32:00Z","title":"SentinelBench: A Benchmark for Long-Running Monitoring Agents","version":2},"reference_index":28,"source":"arxiv_source","source_observed_at":"2026-06-28T06:09:38.698353Z"},"links":{"cited_paper":"/paper/2604.12162","citing_paper":"/paper/2606.05342"},"observation_digest":"sha256:e8cb9a756ac9286d08ee679301e0f4399bbeae8b9162ef0e1b13a80fb5e51248","observation_id":"bd08c571-b324-4482-af18-467b6f476f9f","resolution":{"observed_at":"2026-07-02T08:16:48.332342Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2604.12162/citation-record","integrity":"/paper/2604.12162/integrity","json":"/paper/2604.12162/citation-record.json","paper":"/paper/2604.12162"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2603.01203","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-01T20:06:12.329681Z","title":"assignment","venue":null,"work_id":"cdf24e36-fd17-49ac-bacd-3bb4002bc436","year":2026},"citing_paper":{"arxiv_id":"2604.12162","last_updated":"2026-04-14T00:43:20Z","snapshot_observed_at":"2026-07-31T20:58:59.645474Z","submitted_at":"2026-04-14T00:43:20Z","title":"AlphaEval: Evaluating Agents in Production","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-10T16:30:51.886471Z"},"links":{"citing_paper":"/paper/2604.12162"},"observation_digest":"sha256:63762a4ac4804c29933ff296c55b2ccc7acc30683823e5d56e5c09c431e46d91","observation_id":"c71a63eb-01d1-4d29-af89-fcc9c93579a7","resolution":{"observed_at":"2026-05-11T08:45:58.840459Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.12516","last_updated":"2025-04-16T22:27:45Z","snapshot_observed_at":"2026-08-03T00:43:33.338074Z","submitted_at":"2025-04-16T22:27:45Z","title":"BrowseComp: A Simple Yet Challenging Benchmark for Browsing Agents","version":1},"cited_work":{"arxiv_id":"2504.12516","doi":"10.48550/arxiv.2504.12516","metadata_source":"pith","pith_arxiv_id":"2504.12516","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"BrowseComp: A Simple Yet Challenging Benchmark for Browsing Agents","venue":"cs.CL","work_id":"25adb508-d97c-49d6-ae43-7a70c2478a34","year":2025},"citing_paper":{"arxiv_id":"2604.12162","last_updated":"2026-04-14T00:43:20Z","snapshot_observed_at":"2026-07-31T20:58:59.645474Z","submitted_at":"2026-04-14T00:43:20Z","title":"AlphaEval: Evaluating Agents in Production","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-10T16:30:51.886471Z"},"links":{"cited_paper":"/paper/2504.12516","citing_paper":"/paper/2604.12162"},"observation_digest":"sha256:2c69136fb5575452fb7baf0503519e6f8bc59e6d8b070f4f274d9f781cf6d48b","observation_id":"4ee44383-207c-46ed-8d0a-5c6610146b1f","resolution":{"observed_at":"2026-05-12T07:44:32.193354Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2602.16662","last_updated":"2026-07-02T17:08:21Z","snapshot_observed_at":"2026-08-03T14:15:59.440299Z","submitted_at":"2026-02-18T18:02:51Z","title":"Evaluating Collective Behaviour of Hundreds of LLM Agents","version":2},"cited_work":{"arxiv_id":"2602.16662","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2602.16662","snapshot_observed_at":"2026-07-03T02:17:35.937490Z","title":"Evaluating collective behaviour of hundreds of llm agents","venue":null,"work_id":"d7ed7afe-1fa2-4833-a271-e3e317ede29a","year":2026},"citing_paper":{"arxiv_id":"2604.12162","last_updated":"2026-04-14T00:43:20Z","snapshot_observed_at":"2026-07-31T20:58:59.645474Z","submitted_at":"2026-04-14T00:43:20Z","title":"AlphaEval: Evaluating Agents in Production","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-10T16:30:51.886471Z"},"links":{"cited_paper":"/paper/2602.16662","citing_paper":"/paper/2604.12162"},"observation_digest":"sha256:6ebcb5d022cb38b0f7c28c76ebe4d69eda379328dd4a0aa2b0c8793c732bb70c","observation_id":"dc2d78a1-692c-445e-be93-18b5d2e7cc07","resolution":{"observed_at":"2026-07-03T02:17:35.937490Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.13651","last_updated":"2025-06-16T16:16:14Z","snapshot_observed_at":"2026-08-06T16:42:38.221057Z","submitted_at":"2025-06-16T16:16:14Z","title":"xbench: Tracking Agents Productivity Scaling with Profession-Aligned Real-World Evaluations","version":1},"cited_work":{"arxiv_id":"2506.13651","doi":"10.48550/arxiv.2506.13651","metadata_source":"pith","pith_arxiv_id":"2506.13651","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"xbench: Tracking agents productivity scaling with profession-aligned real-world evaluations","venue":"cs.LG","work_id":"b7f0bdf6-3821-4735-b55e-be58f6ef326d","year":2025},"citing_paper":{"arxiv_id":"2604.12162","last_updated":"2026-04-14T00:43:20Z","snapshot_observed_at":"2026-07-31T20:58:59.645474Z","submitted_at":"2026-04-14T00:43:20Z","title":"AlphaEval: Evaluating Agents in Production","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-10T16:30:51.886471Z"},"links":{"cited_paper":"/paper/2506.13651","citing_paper":"/paper/2604.12162"},"observation_digest":"sha256:186b25455bef1cd9b3c8e994fb180c656776b9f3db27501f76611aa4655598cf","observation_id":"eaaccacc-27fb-4488-b716-c5d265e77e37","resolution":{"observed_at":"2026-05-11T08:45:58.832941Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2602.14721","doi":"10.48550/arxiv.2602.14721","metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Webworld: A large-scale world model for web agent training","venue":"Open MIND","work_id":"9a41f84f-9a89-4bcf-8b7c-2d9631db5fb2","year":2026},"citing_paper":{"arxiv_id":"2604.12162","last_updated":"2026-04-14T00:43:20Z","snapshot_observed_at":"2026-07-31T20:58:59.645474Z","submitted_at":"2026-04-14T00:43:20Z","title":"AlphaEval: Evaluating Agents in Production","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-05-10T16:30:51.886471Z"},"links":{"citing_paper":"/paper/2604.12162"},"observation_digest":"sha256:78b617222dd23b8734764bd4dd21aa7caa87393e35c8c4bc5b82c19f477555ca","observation_id":"c9169c1d-f4b1-46c4-8379-c9c115584ba8","resolution":{"observed_at":"2026-05-11T08:45:58.856521Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.07972","last_updated":"2024-05-30T08:55:12Z","snapshot_observed_at":"2026-07-06T17:59:00.124173Z","submitted_at":"2024-04-11T17:56:05Z","title":"OSWorld: Benchmarking Multimodal Agents for Open-Ended Tasks in Real Computer Environments","version":2},"cited_work":{"arxiv_id":"2404.07972","doi":"10.48550/arxiv.2404.07972","metadata_source":"pith","pith_arxiv_id":"2404.07972","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"OSWorld: Benchmarking Multimodal Agents for Open-Ended Tasks in Real Computer Environments","venue":"cs.AI","work_id":"793d9419-734d-45fe-9f51-d4c5a3a57cf8","year":2024},"citing_paper":{"arxiv_id":"2604.12162","last_updated":"2026-04-14T00:43:20Z","snapshot_observed_at":"2026-07-31T20:58:59.645474Z","submitted_at":"2026-04-14T00:43:20Z","title":"AlphaEval: Evaluating Agents in Production","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-05-10T16:30:51.886471Z"},"links":{"cited_paper":"/paper/2404.07972","citing_paper":"/paper/2604.12162"},"observation_digest":"sha256:56cd699218295fab53ce654c67327bb257214da193336db97bbbd518d5de32cc","observation_id":"ee85e413-d8c0-4d3b-8823-d78918b9f5f4","resolution":{"observed_at":"2026-05-13T01:19:32.642608Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-05-19T22:22:21.817655+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-19T22:22:21.817655+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2412.14161","last_updated":"2025-09-10T08:35:19Z","snapshot_observed_at":"2026-08-01T16:27:28.241667Z","submitted_at":"2024-12-18T18:55:40Z","title":"TheAgentCompany: Benchmarking LLM Agents on Consequential Real World Tasks","version":3},"cited_work":{"arxiv_id":"2412.14161","doi":"10.48550/arxiv.2412.14161","metadata_source":"pith","pith_arxiv_id":"2412.14161","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"TheAgentCompany: Benchmarking LLM Agents on Consequential Real World Tasks","venue":"cs.CL","work_id":"8ba3cce8-4fc7-4286-9bae-513243ed4e6e","year":2024},"citing_paper":{"arxiv_id":"2604.12162","last_updated":"2026-04-14T00:43:20Z","snapshot_observed_at":"2026-07-31T20:58:59.645474Z","submitted_at":"2026-04-14T00:43:20Z","title":"AlphaEval: Evaluating Agents in Production","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-05-10T16:30:51.886471Z"},"links":{"cited_paper":"/paper/2412.14161","citing_paper":"/paper/2604.12162"},"observation_digest":"sha256:5501228899b1a0502fc376fefb602495274a732d700a5099b1404336db2f11f7","observation_id":"6f6ff325-28bf-461d-8b22-c7bf70e4c284","resolution":{"observed_at":"2026-05-14T22:39:31.128039Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.03859","last_updated":"2024-10-04T18:48:58Z","snapshot_observed_at":"2026-07-06T19:28:08.374354Z","submitted_at":"2024-10-04T18:48:58Z","title":"SWE-bench Multimodal: Do AI Systems Generalize to Visual Software Domains?","version":1},"cited_work":{"arxiv_id":"2410.03859","doi":"10.48550/arxiv.2410.03859","metadata_source":"arxiv_reference","pith_arxiv_id":"2410.03859","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv.org","venue":"arXiv (Cornell University)","work_id":"633853a8-4945-471e-8a0f-e3fd7fd02f77","year":2024},"citing_paper":{"arxiv_id":"2604.12162","last_updated":"2026-04-14T00:43:20Z","snapshot_observed_at":"2026-07-31T20:58:59.645474Z","submitted_at":"2026-04-14T00:43:20Z","title":"AlphaEval: Evaluating Agents in Production","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-05-10T16:30:51.886471Z"},"links":{"cited_paper":"/paper/2410.03859","citing_paper":"/paper/2604.12162"},"observation_digest":"sha256:88cfb6d1bff6a068dacbe4fbcd7659482f75cad72f737e96c0d8e51047821076","observation_id":"6e31f8ef-6b5a-44c4-8f5d-6d244db49fe9","resolution":{"observed_at":"2026-05-11T08:45:58.877213Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2603.07980","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-04T20:00:08.402369Z","title":"$OneMillion-Bench: How far are language agents from human experts?arXiv preprint arXiv:2603.07980","venue":null,"work_id":"8519ff23-797d-41f2-834f-3bd0f249949d","year":2026},"citing_paper":{"arxiv_id":"2604.12162","last_updated":"2026-04-14T00:43:20Z","snapshot_observed_at":"2026-07-31T20:58:59.645474Z","submitted_at":"2026-04-14T00:43:20Z","title":"AlphaEval: Evaluating Agents in Production","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-05-10T16:30:51.886471Z"},"links":{"citing_paper":"/paper/2604.12162"},"observation_digest":"sha256:9d40017345b6cc7c16d581eea0994ee7f8ef31cb20aa04082ca502dc519626a3","observation_id":"cb0cb7a0-d26c-4ac9-a6af-8138c5ca62f9","resolution":{"observed_at":"2026-05-11T08:45:58.885606Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2406.12045","last_updated":"2024-06-17T19:33:08Z","snapshot_observed_at":"2026-08-02T22:19:29.043854Z","submitted_at":"2024-06-17T19:33:08Z","title":"$\\tau$-bench: A Benchmark for Tool-Agent-User Interaction in Real-World Domains","version":1},"cited_work":{"arxiv_id":"2406.12045","doi":"10.48550/arxiv.2406.12045","metadata_source":"pith","pith_arxiv_id":"2406.12045","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"$\\tau$-bench: A Benchmark for Tool-Agent-User Interaction in Real-World Domains","venue":"cs.AI","work_id":"6a8d8dc4-0cc0-4052-8109-abbcdcd4a962","year":2024},"citing_paper":{"arxiv_id":"2604.12162","last_updated":"2026-04-14T00:43:20Z","snapshot_observed_at":"2026-07-31T20:58:59.645474Z","submitted_at":"2026-04-14T00:43:20Z","title":"AlphaEval: Evaluating Agents in Production","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-05-10T16:30:51.886471Z"},"links":{"cited_paper":"/paper/2406.12045","citing_paper":"/paper/2604.12162"},"observation_digest":"sha256:30043438c85163497dd45c6978e5362ec2ed7f182eef154bbeb03ff376912141","observation_id":"648d44ea-c9dd-4f93-ad56-2609760777da","resolution":{"observed_at":"2026-05-11T08:45:58.844635Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-14T18:20:21.86453+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-14T18:20:21.86453+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.16502","last_updated":"2024-06-13T15:02:39Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-11-27T17:33:21Z","title":"MMMU: A Massive Multi-discipline Multimodal Understanding and Reasoning Benchmark for Expert AGI","version":4},"cited_work":{"arxiv_id":"2311.16502","doi":"10.48550/arxiv.2311.16502","metadata_source":"pith","pith_arxiv_id":"2311.16502","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MMMU: A Massive Multi-discipline Multimodal Understanding and Reasoning Benchmark for Expert AGI","venue":"cs.CL","work_id":"da087b16-ea05-4064-980e-ce1d6e281d49","year":2023},"citing_paper":{"arxiv_id":"2604.12162","last_updated":"2026-04-14T00:43:20Z","snapshot_observed_at":"2026-07-31T20:58:59.645474Z","submitted_at":"2026-04-14T00:43:20Z","title":"AlphaEval: Evaluating Agents in Production","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-05-10T16:30:51.886471Z"},"links":{"cited_paper":"/paper/2311.16502","citing_paper":"/paper/2604.12162"},"observation_digest":"sha256:5926c09c94f281f06d8862e68857f5cc10640e25fc0a75ad77adf892a837a8a9","observation_id":"f9f8fc77-4832-46ce-b469-f8754df43f45","resolution":{"observed_at":"2026-05-15T05:37:42.029646Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.02605","last_updated":"2025-04-03T14:06:17Z","snapshot_observed_at":"2026-07-30T07:56:48.878578Z","submitted_at":"2025-04-03T14:06:17Z","title":"Multi-SWE-bench: A Multilingual Benchmark for Issue Resolving","version":1},"cited_work":{"arxiv_id":"2504.02605","doi":"10.48550/arxiv.2504.02605","metadata_source":"pith","pith_arxiv_id":"2504.02605","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Multi-SWE-bench: A Multilingual Benchmark for Issue Resolving","venue":"cs.SE","work_id":"c66ce635-0931-4807-8300-a45863330d75","year":2025},"citing_paper":{"arxiv_id":"2604.12162","last_updated":"2026-04-14T00:43:20Z","snapshot_observed_at":"2026-07-31T20:58:59.645474Z","submitted_at":"2026-04-14T00:43:20Z","title":"AlphaEval: Evaluating Agents in Production","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-05-10T16:30:51.886471Z"},"links":{"cited_paper":"/paper/2504.02605","citing_paper":"/paper/2604.12162"},"observation_digest":"sha256:2ec3e54d7d1bbfc19cb083c21e996d4b61f0889ca71e9165e8eba65e8ec4aabf","observation_id":"274d648c-9e66-442e-9fcd-957718ce5dec","resolution":{"observed_at":"2026-05-16T06:48:50.578231Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-07-11T01:50:49.442994+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-11T01:50:49.442994+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.08926","last_updated":"2025-04-12T21:26:07Z","snapshot_observed_at":"2026-08-06T12:22:30.341073Z","submitted_at":"2024-08-15T17:23:10Z","title":"Cybench: A Framework for Evaluating Cybersecurity Capabilities and Risks of Language Models","version":4},"cited_work":{"arxiv_id":"2408.08926","doi":null,"metadata_source":"pith","pith_arxiv_id":"2408.08926","snapshot_observed_at":"2026-07-10T18:37:31.168670Z","title":"K., et al","venue":"cs.CR","work_id":"e0f8e335-0df1-4fa1-a820-ae6717e73cc1","year":2024},"citing_paper":{"arxiv_id":"2604.12162","last_updated":"2026-04-14T00:43:20Z","snapshot_observed_at":"2026-07-31T20:58:59.645474Z","submitted_at":"2026-04-14T00:43:20Z","title":"AlphaEval: Evaluating Agents in Production","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-05-10T16:30:51.886471Z"},"links":{"cited_paper":"/paper/2408.08926","citing_paper":"/paper/2604.12162"},"observation_digest":"sha256:aac4353c4415b41e7f941b3af39bd44f12f1c7f8e2f45ac7d67d689a56bfadf2","observation_id":"58f31584-7ea9-41de-ab10-7890120d3c43","resolution":{"observed_at":"2026-05-11T08:45:58.817569Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2505.15216","doi":"10.48550/arxiv.2505.15216","metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Zhang, Joey Ji, Celeste Menders, Riya Dulepet, T","venue":"ArXiv.org","work_id":"13c84db8-c5ad-4f46-880d-9c5e55ddbd8d","year":2025},"citing_paper":{"arxiv_id":"2604.12162","last_updated":"2026-04-14T00:43:20Z","snapshot_observed_at":"2026-07-31T20:58:59.645474Z","submitted_at":"2026-04-14T00:43:20Z","title":"AlphaEval: Evaluating Agents in Production","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-05-10T16:30:51.886471Z"},"links":{"citing_paper":"/paper/2604.12162"},"observation_digest":"sha256:c860ccc2a6139f9e100fbab8fe5341b51c753d6ed231a66c6f6ee898f2feb58d","observation_id":"d4c1823a-3c93-420f-acf8-92088040d1e4","resolution":{"observed_at":"2026-05-11T08:45:58.811047Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2602.12876","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Browsecomp-v3: A visual, vertical, and verifiable benchmark for multimodal browsing agents","venue":null,"work_id":"788e207a-4938-40a7-9a3a-242a642116cf","year":2026},"citing_paper":{"arxiv_id":"2604.12162","last_updated":"2026-04-14T00:43:20Z","snapshot_observed_at":"2026-07-31T20:58:59.645474Z","submitted_at":"2026-04-14T00:43:20Z","title":"AlphaEval: Evaluating Agents in Production","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-05-10T16:30:51.886471Z"},"links":{"citing_paper":"/paper/2604.12162"},"observation_digest":"sha256:97deb8ebf6fd1a57bc76b091f058c66e008517f32a8af31e271153b00dcf4dab","observation_id":"29b4da49-2e8c-4344-8692-ed656bd1e3d2","resolution":{"observed_at":"2026-05-11T08:45:58.779600Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2306.05685","last_updated":"2023-12-24T02:01:34Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-06-09T05:55:52Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","version":4},"cited_work":{"arxiv_id":"2306.05685","doi":"10.1109/4235.797969","metadata_source":"pith","pith_arxiv_id":"2306.05685","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Judging LLM-as-a-Judge with MT-Bench and Chatbot Arena","venue":"cs.CL","work_id":"d0c30cd7-81e1-4159-a87f-f6adca77ff08","year":2023},"citing_paper":{"arxiv_id":"2604.12162","last_updated":"2026-04-14T00:43:20Z","snapshot_observed_at":"2026-07-31T20:58:59.645474Z","submitted_at":"2026-04-14T00:43:20Z","title":"AlphaEval: Evaluating Agents in Production","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-05-10T16:30:51.886471Z"},"links":{"cited_paper":"/paper/2306.05685","citing_paper":"/paper/2604.12162"},"observation_digest":"sha256:8c032f090d3fee75bcfb7365e40f646dcd80b002074f8e1c0577eb11b8222190","observation_id":"0d493818-4289-42c2-b83e-756e62e965f3","resolution":{"observed_at":"2026-05-11T08:45:58.821265Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2307.13854","last_updated":"2024-04-16T15:13:18Z","snapshot_observed_at":"2026-08-06T12:47:01.809185Z","submitted_at":"2023-07-25T22:59:32Z","title":"WebArena: A Realistic Web Environment for Building Autonomous Agents","version":4},"cited_work":{"arxiv_id":"2307.13854","doi":"10.48550/arxiv.2307.13854","metadata_source":"pith","pith_arxiv_id":"2307.13854","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"WebArena: A Realistic Web Environment for Building Autonomous Agents","venue":"cs.AI","work_id":"7058ffd2-a339-4102-89eb-248eeb074652","year":2023},"citing_paper":{"arxiv_id":"2604.12162","last_updated":"2026-04-14T00:43:20Z","snapshot_observed_at":"2026-07-31T20:58:59.645474Z","submitted_at":"2026-04-14T00:43:20Z","title":"AlphaEval: Evaluating Agents in Production","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-05-10T16:30:51.886471Z"},"links":{"cited_paper":"/paper/2307.13854","citing_paper":"/paper/2604.12162"},"observation_digest":"sha256:046c714dd79e1907dcec4d18882a9be4b344b814b65f2400e4438cda5bba2b0c","observation_id":"d97ccce5-150f-411a-a200-04f6515d7e9b","resolution":{"observed_at":"2026-05-11T08:45:58.801295Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-05-20T18:52:18.85917+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-20T18:52:18.85917+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.10934","last_updated":"2024-10-16T17:54:12Z","snapshot_observed_at":"2026-08-05T11:14:45.116658Z","submitted_at":"2024-10-14T17:57:02Z","title":"Agent-as-a-Judge: Evaluate Agents with Agents","version":2},"cited_work":{"arxiv_id":"2410.10934","doi":null,"metadata_source":"pith","pith_arxiv_id":"2410.10934","snapshot_observed_at":"2026-07-10T15:07:19.644728Z","title":"arXiv preprint 2410.10934","venue":"cs.AI","work_id":"2e127924-0ccf-418b-9de5-0daa4fb9ad9b","year":2024},"citing_paper":{"arxiv_id":"2604.12162","last_updated":"2026-04-14T00:43:20Z","snapshot_observed_at":"2026-07-31T20:58:59.645474Z","submitted_at":"2026-04-14T00:43:20Z","title":"AlphaEval: Evaluating Agents in Production","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-05-10T16:30:51.886471Z"},"links":{"cited_paper":"/paper/2410.10934","citing_paper":"/paper/2604.12162"},"observation_digest":"sha256:f25ee8ac2097f41bb725cf519409343e133290eb4bb4b20f72f5e22866e9f67d","observation_id":"33fa85b9-3db3-4e29-814f-ffbaeec43fce","resolution":{"observed_at":"2026-05-11T08:45:58.864563Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"d1516d84-fa44-4549-99a8-b0c0c9aaf436","year":null},"citing_paper":{"arxiv_id":"2604.12162","last_updated":"2026-04-14T00:43:20Z","snapshot_observed_at":"2026-07-31T20:58:59.645474Z","submitted_at":"2026-04-14T00:43:20Z","title":"AlphaEval: Evaluating Agents in Production","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-05-10T16:30:51.886471Z"},"links":{"citing_paper":"/paper/2604.12162"},"observation_digest":"sha256:86d57257a685ca9e097c0df4369a91c06d9af54eb870f5243e841ba411515073","observation_id":"1ac7c2af-3768-4470-bb5f-00a94afaa875","resolution":{"observed_at":"2026-05-17T14:59:48.843258Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"29056675-867d-47e3-85e9-600f9a454597","year":null},"citing_paper":{"arxiv_id":"2604.12162","last_updated":"2026-04-14T00:43:20Z","snapshot_observed_at":"2026-07-31T20:58:59.645474Z","submitted_at":"2026-04-14T00:43:20Z","title":"AlphaEval: Evaluating Agents in Production","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-05-10T16:30:51.886471Z"},"links":{"citing_paper":"/paper/2604.12162"},"observation_digest":"sha256:8aaf40b418d1089a91ff8ddec875ccd8ddda42336e901e53ef3b22ec1c717354","observation_id":"3ea5f684-78a2-4f75-8530-27b5a8ef1365","resolution":{"observed_at":"2026-05-17T14:59:48.840636Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"no criteria","venue":null,"work_id":"9373cbba-6fd7-4ea2-9d10-e67a1eaeae6a","year":null},"citing_paper":{"arxiv_id":"2604.12162","last_updated":"2026-04-14T00:43:20Z","snapshot_observed_at":"2026-07-31T20:58:59.645474Z","submitted_at":"2026-04-14T00:43:20Z","title":"AlphaEval: Evaluating Agents in Production","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-05-10T16:30:51.886471Z"},"links":{"citing_paper":"/paper/2604.12162"},"observation_digest":"sha256:f0208628d5c57917e98fe49e24e029afa02d6ed588138d9c903c3a3629e863ea","observation_id":"826eda7f-4d7b-445b-b426-832aa1e3d216","resolution":{"observed_at":"2026-05-17T14:59:48.837685Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-06T06:34:29.942622+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2604.12162","last_updated":"2026-04-14T00:43:20Z","latest_version":1,"primary_category":"cs.CL","snapshot_observed_at":"2026-07-31T20:58:59.645474Z","submitted_at":"2026-04-14T00:43:20Z","title":"AlphaEval: Evaluating Agents in Production"},"reference_resolution":{"displayed":21,"state_counts":{"malformed_identifier":0,"metadata_mismatch":1,"parse_uncertain":0,"unresolved":2,"verified_exact":17,"verified_fuzzy":1},"total_outbound_references":21},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-06T06:34:29.942622+00:00","source":"crossref"},{"observed_at":"2026-08-06T06:34:23.284952+00:00","source":"retraction_watch"}],"thesis":"As of 6 August 2026, this Paper Citation Record lists 21 of 21 outbound references and 1 inbound Pith citation observation for arXiv:2604.12162."}