{"as_of":"2026-08-21T00:30:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:a61f2f3c9faa9ee9867dc9542e80280beefc97dd4873b31f286e09baf39bfee4","coverage":[{"denominator":118,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":100,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-06T20:24:29.338123Z","state":"measured"},{"denominator":137,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":137,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-20T06:33:59.587034+00:00","state":"measured"},{"denominator":37,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":37,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-15T14:19:14.004821Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"arxiv_reference","source_observed_at":"2026-07-03T22:08:59.930877Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2507.02825","last_updated":"2025-08-07T06:58:08Z","snapshot_observed_at":"2026-08-17T22:49:53.742436Z","submitted_at":"2025-07-03T17:35:31Z","title":"Establishing Best Practices for Building Rigorous Agentic Benchmarks","version":5},"cited_work":{"arxiv_id":"2507.02825","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2507.02825","snapshot_observed_at":"2026-07-03T22:08:59.930877Z","title":"Establishing best practices for building rigorous agentic benchmarks","venue":null,"work_id":"981c0760-99e5-4175-bca8-6be3ed159f5b","year":2025},"citing_paper":{"arxiv_id":"2601.11868","last_updated":"2026-01-17T01:29:30Z","snapshot_observed_at":"2026-08-20T02:32:10.164015Z","submitted_at":"2026-01-17T01:29:30Z","title":"Terminal-Bench: Benchmarking Agents on Hard, Realistic Tasks in Command Line Interfaces","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-05-11T03:37:07.841385Z"},"links":{"cited_paper":"/paper/2507.02825","citing_paper":"/paper/2601.11868"},"observation_digest":"sha256:7986b383546b5cd4c793c48ac156f64cde25a6a854f78dbcc54b426a6741de67","observation_id":"66ecafb5-6e96-4e31-a056-9461b2744708","resolution":{"observed_at":"2026-05-11T03:37:07.992049Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.02825","last_updated":"2025-08-07T06:58:08Z","snapshot_observed_at":"2026-08-17T22:49:53.742436Z","submitted_at":"2025-07-03T17:35:31Z","title":"Establishing Best Practices for Building Rigorous Agentic Benchmarks","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2507.02825","snapshot_observed_at":"2026-07-13T15:55:53.399860Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2603.29139","last_updated":"2026-08-09T04:03:34Z","snapshot_observed_at":"2026-08-16T05:17:14.711870Z","submitted_at":"2026-03-31T01:41:28Z","title":"SciVisAgentBench: A Benchmark for Evaluating Scientific Data Analysis and Visualization Agents","version":2},"reference_index":80,"source":"pdf_text","source_observed_at":"2026-07-13T15:55:53.399860Z"},"links":{"cited_paper":"/paper/2507.02825","citing_paper":"/paper/2603.29139"},"observation_digest":"sha256:15dc9a56dd874a2bd818ed05fe0aeebee818add4226f90b05b736d816e0f811e","observation_id":"a6403359-918a-4741-a1f5-4ba21eaec78d","resolution":{"observed_at":"2026-07-13T15:55:53.399860Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2507.02825","last_updated":"2025-08-07T06:58:08Z","snapshot_observed_at":"2026-08-17T22:49:53.742436Z","submitted_at":"2025-07-03T17:35:31Z","title":"Establishing Best Practices for Building Rigorous Agentic Benchmarks","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2507.02825","snapshot_observed_at":"2026-08-02T17:09:18.219867Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2603.29139","last_updated":"2026-08-09T04:03:34Z","snapshot_observed_at":"2026-08-16T05:17:14.711870Z","submitted_at":"2026-03-31T01:41:28Z","title":"SciVisAgentBench: A Benchmark for Evaluating Scientific Data Analysis and Visualization Agents","version":3},"reference_index":72,"source":"pdf_text","source_observed_at":"2026-08-02T17:09:18.219867Z"},"links":{"cited_paper":"/paper/2507.02825","citing_paper":"/paper/2603.29139"},"observation_digest":"sha256:300e47f2c0be15b0ffbe9e99a96623d516657f3ee741a71eadfb951160197a3e","observation_id":"cb12bd46-df97-4812-ba29-41819a3a7b8b","resolution":{"observed_at":"2026-08-02T17:09:18.219867Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2507.02825","last_updated":"2025-08-07T06:58:08Z","snapshot_observed_at":"2026-08-17T22:49:53.742436Z","submitted_at":"2025-07-03T17:35:31Z","title":"Establishing Best Practices for Building Rigorous Agentic Benchmarks","version":5},"cited_work":{"arxiv_id":"2507.02825","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2507.02825","snapshot_observed_at":"2026-07-03T22:08:59.930877Z","title":"Establishing best practices for building rigorous agentic benchmarks","venue":null,"work_id":"981c0760-99e5-4175-bca8-6be3ed159f5b","year":2025},"citing_paper":{"arxiv_id":"2604.05229","last_updated":"2026-04-06T22:49:28Z","snapshot_observed_at":"2026-08-12T17:47:23.301971Z","submitted_at":"2026-04-06T22:49:28Z","title":"From Governance Norms to Enforceable Controls: A Layered Translation Method for Runtime Guardrails in Agentic AI","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-05-10T18:47:09.928166Z"},"links":{"cited_paper":"/paper/2507.02825","citing_paper":"/paper/2604.05229"},"observation_digest":"sha256:b6a08f116519a15c3698e3dfdb7ef3a171bbd8f3cd310e7fd83aa7b53fb1410f","observation_id":"65b39ab3-2771-400f-8549-74a87ee7986f","resolution":{"observed_at":"2026-05-10T23:55:51.819572Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.02825","last_updated":"2025-08-07T06:58:08Z","snapshot_observed_at":"2026-08-17T22:49:53.742436Z","submitted_at":"2025-07-03T17:35:31Z","title":"Establishing Best Practices for Building Rigorous Agentic Benchmarks","version":5},"cited_work":{"arxiv_id":"2507.02825","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2507.02825","snapshot_observed_at":"2026-07-03T22:08:59.930877Z","title":"Establishing best practices for building rigorous agentic benchmarks","venue":null,"work_id":"981c0760-99e5-4175-bca8-6be3ed159f5b","year":2025},"citing_paper":{"arxiv_id":"2604.19818","last_updated":"2026-04-18T20:28:26Z","snapshot_observed_at":"2026-08-16T05:11:27.917300Z","submitted_at":"2026-04-18T20:28:26Z","title":"Beyond Task Success: An Evidence-Synthesis Framework for Evaluating, Governing, and Orchestrating Agentic AI","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-10T06:03:19.805349Z"},"links":{"cited_paper":"/paper/2507.02825","citing_paper":"/paper/2604.19818"},"observation_digest":"sha256:ea3c3eaaeebb5cd0a9c1d0661ae642fe7a970e9a6b5d2cee17c1ace6079b88a8","observation_id":"c1e2136b-abc8-4404-9894-f3aec855f5b5","resolution":{"observed_at":"2026-05-10T06:06:18.917983Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.02825","last_updated":"2025-08-07T06:58:08Z","snapshot_observed_at":"2026-08-17T22:49:53.742436Z","submitted_at":"2025-07-03T17:35:31Z","title":"Establishing Best Practices for Building Rigorous Agentic Benchmarks","version":5},"cited_work":{"arxiv_id":"2507.02825","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2507.02825","snapshot_observed_at":"2026-07-03T22:08:59.930877Z","title":"Establishing best practices for building rigorous agentic benchmarks","venue":null,"work_id":"981c0760-99e5-4175-bca8-6be3ed159f5b","year":2025},"citing_paper":{"arxiv_id":"2605.00927","last_updated":"2026-04-30T19:22:44Z","snapshot_observed_at":"2026-08-17T19:16:30.905354Z","submitted_at":"2026-04-30T19:22:44Z","title":"BioVeil MATRIX: Uncovering and categorizing vulnerabilities of agentic biological AI scientists","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-05-09T20:03:59.215134Z"},"links":{"cited_paper":"/paper/2507.02825","citing_paper":"/paper/2605.00927"},"observation_digest":"sha256:490b105579a49c2d0653445bebe1f0681746cd3fcf1ab28881875473dc83e6ac","observation_id":"d47897a7-e545-4666-9bff-02f2187ace23","resolution":{"observed_at":"2026-05-11T15:26:08.439353Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.02825","last_updated":"2025-08-07T06:58:08Z","snapshot_observed_at":"2026-08-17T22:49:53.742436Z","submitted_at":"2025-07-03T17:35:31Z","title":"Establishing Best Practices for Building Rigorous Agentic Benchmarks","version":5},"cited_work":{"arxiv_id":"2507.02825","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2507.02825","snapshot_observed_at":"2026-07-03T22:08:59.930877Z","title":"Establishing best practices for building rigorous agentic benchmarks","venue":null,"work_id":"981c0760-99e5-4175-bca8-6be3ed159f5b","year":2025},"citing_paper":{"arxiv_id":"2605.07073","last_updated":"2026-05-08T00:48:45Z","snapshot_observed_at":"2026-08-12T20:58:22.795261Z","submitted_at":"2026-05-08T00:48:45Z","title":"TeamBench: Evaluating Agent Coordination under Enforced Role Separation","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-05-11T00:55:51.358828Z"},"links":{"cited_paper":"/paper/2507.02825","citing_paper":"/paper/2605.07073"},"observation_digest":"sha256:04bd71d8458bbebfdf7c33c34ee3f5c482a295bf81542c0f33087432b03817e4","observation_id":"0890bdcf-531d-43e4-ab67-24038f357d5b","resolution":{"observed_at":"2026-05-11T05:00:55.552635Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.02825","last_updated":"2025-08-07T06:58:08Z","snapshot_observed_at":"2026-08-17T22:49:53.742436Z","submitted_at":"2025-07-03T17:35:31Z","title":"Establishing Best Practices for Building Rigorous Agentic Benchmarks","version":5},"cited_work":{"arxiv_id":"2507.02825","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2507.02825","snapshot_observed_at":"2026-07-03T22:08:59.930877Z","title":"Establishing best practices for building rigorous agentic benchmarks","venue":null,"work_id":"981c0760-99e5-4175-bca8-6be3ed159f5b","year":2025},"citing_paper":{"arxiv_id":"2605.07161","last_updated":"2026-07-30T18:34:51Z","snapshot_observed_at":"2026-08-13T20:31:34.793657Z","submitted_at":"2026-05-08T02:47:07Z","title":"SREGym: A Live Benchmark for AI SRE Agents with High-Fidelity Failure Scenarios","version":1},"reference_index":94,"source":"pdf_text","source_observed_at":"2026-05-11T02:20:32.528550Z"},"links":{"cited_paper":"/paper/2507.02825","citing_paper":"/paper/2605.07161"},"observation_digest":"sha256:bb6749df8acb5b910aaef4688907af703006f848e8e12936df6c1c5822ee893f","observation_id":"33d5ddc6-5eda-4558-8d43-3812ad576fee","resolution":{"observed_at":"2026-05-11T02:20:54.825579Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.02825","last_updated":"2025-08-07T06:58:08Z","snapshot_observed_at":"2026-08-17T22:49:53.742436Z","submitted_at":"2025-07-03T17:35:31Z","title":"Establishing Best Practices for Building Rigorous Agentic Benchmarks","version":5},"cited_work":{"arxiv_id":"2507.02825","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2507.02825","snapshot_observed_at":"2026-07-03T22:08:59.930877Z","title":"Establishing best practices for building rigorous agentic benchmarks","venue":null,"work_id":"981c0760-99e5-4175-bca8-6be3ed159f5b","year":2025},"citing_paper":{"arxiv_id":"2605.07161","last_updated":"2026-07-30T18:34:51Z","snapshot_observed_at":"2026-08-13T20:31:34.793657Z","submitted_at":"2026-05-08T02:47:07Z","title":"SREGym: A Live Benchmark for AI SRE Agents with High-Fidelity Failure Scenarios","version":2},"reference_index":92,"source":"pdf_text","source_observed_at":"2026-05-14T21:49:16.239350Z"},"links":{"cited_paper":"/paper/2507.02825","citing_paper":"/paper/2605.07161"},"observation_digest":"sha256:fbf49c3b627061880d9b88e85f004e25a3b39d2f9fec6c5b4ab9b8599419d93e","observation_id":"20d96c9f-295e-4276-9f73-5c280b826463","resolution":{"observed_at":"2026-05-14T21:49:29.231213Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.02825","last_updated":"2025-08-07T06:58:08Z","snapshot_observed_at":"2026-08-17T22:49:53.742436Z","submitted_at":"2025-07-03T17:35:31Z","title":"Establishing Best Practices for Building Rigorous Agentic Benchmarks","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2507.02825","snapshot_observed_at":"2026-08-03T00:18:46.882265Z","title":"mitigation","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2605.07161","last_updated":"2026-07-30T18:34:51Z","snapshot_observed_at":"2026-08-13T20:31:34.793657Z","submitted_at":"2026-05-08T02:47:07Z","title":"SREGym: A Live Benchmark for AI SRE Agents with High-Fidelity Failure Scenarios","version":3},"reference_index":92,"source":"pdf_text","source_observed_at":"2026-08-03T00:18:46.882265Z"},"links":{"cited_paper":"/paper/2507.02825","citing_paper":"/paper/2605.07161"},"observation_digest":"sha256:e89ce99a26b2a9732f43f1f97d48e52b1f7ab3a606ccd6c4f8f2534f0993d13c","observation_id":"852088d3-e55a-41cd-be32-2580980085b6","resolution":{"observed_at":"2026-08-03T00:18:46.882265Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2507.02825","last_updated":"2025-08-07T06:58:08Z","snapshot_observed_at":"2026-08-17T22:49:53.742436Z","submitted_at":"2025-07-03T17:35:31Z","title":"Establishing Best Practices for Building Rigorous Agentic Benchmarks","version":5},"cited_work":{"arxiv_id":"2507.02825","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2507.02825","snapshot_observed_at":"2026-07-03T22:08:59.930877Z","title":"Establishing best practices for building rigorous agentic benchmarks","venue":null,"work_id":"981c0760-99e5-4175-bca8-6be3ed159f5b","year":2025},"citing_paper":{"arxiv_id":"2605.08468","last_updated":"2026-05-08T20:39:32Z","snapshot_observed_at":"2026-08-02T16:12:56.247576Z","submitted_at":"2026-05-08T20:39:32Z","title":"PYTHALAB-MERA: Validation-Grounded Memory, Retrieval, and Acceptance Control for Frozen-LLM Coding Agents","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-05-12T01:12:37.970638Z"},"links":{"cited_paper":"/paper/2507.02825","citing_paper":"/paper/2605.08468"},"observation_digest":"sha256:56a35ff2f8ecea56c245169157be2392d327236d777068cf6a58d4b1e3f399fc","observation_id":"de977e8c-5c6c-438f-9eed-76228711bbfc","resolution":{"observed_at":"2026-05-12T08:21:26.103689Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.02825","last_updated":"2025-08-07T06:58:08Z","snapshot_observed_at":"2026-08-17T22:49:53.742436Z","submitted_at":"2025-07-03T17:35:31Z","title":"Establishing Best Practices for Building Rigorous Agentic Benchmarks","version":5},"cited_work":{"arxiv_id":"2507.02825","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2507.02825","snapshot_observed_at":"2026-07-03T22:08:59.930877Z","title":"Establishing best practices for building rigorous agentic benchmarks","venue":null,"work_id":"981c0760-99e5-4175-bca8-6be3ed159f5b","year":2025},"citing_paper":{"arxiv_id":"2605.10448","last_updated":"2026-05-11T12:20:15Z","snapshot_observed_at":"2026-08-20T16:41:30.569480Z","submitted_at":"2026-05-11T12:20:15Z","title":"Can Agent Benchmarks Support Their Scores? Evidence-Supported Bounds for Interactive-Agent Evaluation","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-05-12T05:05:55.592359Z"},"links":{"cited_paper":"/paper/2507.02825","citing_paper":"/paper/2605.10448"},"observation_digest":"sha256:b6601f33a4d8b92e972ff0b07e313c23c6d02a94f195ae533058129c2b9fb9f8","observation_id":"927faacd-9882-41fd-a98c-8bd197aaf590","resolution":{"observed_at":"2026-05-12T05:41:24.002297Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.02825","last_updated":"2025-08-07T06:58:08Z","snapshot_observed_at":"2026-08-17T22:49:53.742436Z","submitted_at":"2025-07-03T17:35:31Z","title":"Establishing Best Practices for Building Rigorous Agentic Benchmarks","version":5},"cited_work":{"arxiv_id":"2507.02825","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2507.02825","snapshot_observed_at":"2026-07-03T22:08:59.930877Z","title":"Establishing best practices for building rigorous agentic benchmarks","venue":null,"work_id":"981c0760-99e5-4175-bca8-6be3ed159f5b","year":2025},"citing_paper":{"arxiv_id":"2605.12673","last_updated":"2026-05-12T19:22:45Z","snapshot_observed_at":"2026-08-15T14:26:18.460325Z","submitted_at":"2026-05-12T19:22:45Z","title":"Do Androids Dream of Breaking the Game? Systematically Auditing AI Agent Benchmarks with BenchJack","version":1},"reference_index":68,"source":"pdf_text","source_observed_at":"2026-05-14T20:31:50.043920Z"},"links":{"cited_paper":"/paper/2507.02825","citing_paper":"/paper/2605.12673"},"observation_digest":"sha256:d27f07829ef48eb1b27fa03038b45ac9491990575272c406becc6647d2f580c2","observation_id":"54fc18c7-13b3-4f01-ba8a-8a9115d6c13d","resolution":{"observed_at":"2026-05-14T20:32:56.775266Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.02825","last_updated":"2025-08-07T06:58:08Z","snapshot_observed_at":"2026-08-17T22:49:53.742436Z","submitted_at":"2025-07-03T17:35:31Z","title":"Establishing Best Practices for Building Rigorous Agentic Benchmarks","version":5},"cited_work":{"arxiv_id":"2507.02825","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2507.02825","snapshot_observed_at":"2026-07-03T22:08:59.930877Z","title":"Establishing best practices for building rigorous agentic benchmarks","venue":null,"work_id":"981c0760-99e5-4175-bca8-6be3ed159f5b","year":2025},"citing_paper":{"arxiv_id":"2605.13139","last_updated":"2026-05-13T08:05:16Z","snapshot_observed_at":"2026-08-14T08:57:47.657620Z","submitted_at":"2026-05-13T08:05:16Z","title":"SWE-Cycle: Benchmarking Code Agents across the Complete Issue Resolution Cycle","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-05-14T18:34:39.997353Z"},"links":{"cited_paper":"/paper/2507.02825","citing_paper":"/paper/2605.13139"},"observation_digest":"sha256:cfbdc62a5d83e557ac4bf2ea9f176da4c0aba8adae930f0ec6896a5fdcc3e425","observation_id":"439b1478-b2b6-49fe-b683-513b6b3c67d9","resolution":{"observed_at":"2026-05-14T18:37:35.612990Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.02825","last_updated":"2025-08-07T06:58:08Z","snapshot_observed_at":"2026-08-17T22:49:53.742436Z","submitted_at":"2025-07-03T17:35:31Z","title":"Establishing Best Practices for Building Rigorous Agentic Benchmarks","version":5},"cited_work":{"arxiv_id":"2507.02825","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2507.02825","snapshot_observed_at":"2026-07-03T22:08:59.930877Z","title":"Establishing best practices for building rigorous agentic benchmarks","venue":null,"work_id":"981c0760-99e5-4175-bca8-6be3ed159f5b","year":2025},"citing_paper":{"arxiv_id":"2605.13950","last_updated":"2026-05-13T18:00:00Z","snapshot_observed_at":"2026-08-20T19:27:12.997792Z","submitted_at":"2026-05-13T18:00:00Z","title":"Collider-Bench: Benchmarking AI Agents with Particle Physics Analysis Reproduction","version":1},"reference_index":36,"source":"arxiv_source","source_observed_at":"2026-05-15T06:04:03.605898Z"},"links":{"cited_paper":"/paper/2507.02825","citing_paper":"/paper/2605.13950"},"observation_digest":"sha256:52440c1408c5be41486dc7cea32071f0c13834d49ea56445b759a711f3682a0d","observation_id":"d7d74fb1-4748-4cab-b90e-c60c8b55a756","resolution":{"observed_at":"2026-05-15T06:05:06.699293Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.02825","last_updated":"2025-08-07T06:58:08Z","snapshot_observed_at":"2026-08-17T22:49:53.742436Z","submitted_at":"2025-07-03T17:35:31Z","title":"Establishing Best Practices for Building Rigorous Agentic Benchmarks","version":5},"cited_work":{"arxiv_id":"2507.02825","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2507.02825","snapshot_observed_at":"2026-07-03T22:08:59.930877Z","title":"Establishing best practices for building rigorous agentic benchmarks","venue":null,"work_id":"981c0760-99e5-4175-bca8-6be3ed159f5b","year":2025},"citing_paper":{"arxiv_id":"2605.22564","last_updated":"2026-05-21T14:45:02Z","snapshot_observed_at":"2026-08-17T01:56:18.979421Z","submitted_at":"2026-05-21T14:45:02Z","title":"SynAE: A Framework for Measuring the Quality of Synthetic Data for Tool-Calling Agent Evaluations","version":1},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-05-22T06:25:06.024542Z"},"links":{"cited_paper":"/paper/2507.02825","citing_paper":"/paper/2605.22564"},"observation_digest":"sha256:e16b6876d9d2a4f5b6f8f7b4e7e695c359dea4ba74559c82aa2b4f744587a4fa","observation_id":"b70d3b7a-b3ad-4582-9902-4c5ae2cc26c9","resolution":{"observed_at":"2026-05-22T06:26:09.995466Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.02825","last_updated":"2025-08-07T06:58:08Z","snapshot_observed_at":"2026-08-17T22:49:53.742436Z","submitted_at":"2025-07-03T17:35:31Z","title":"Establishing Best Practices for Building Rigorous Agentic Benchmarks","version":5},"cited_work":{"arxiv_id":"2507.02825","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2507.02825","snapshot_observed_at":"2026-07-03T22:08:59.930877Z","title":"Establishing best practices for building rigorous agentic benchmarks","venue":null,"work_id":"981c0760-99e5-4175-bca8-6be3ed159f5b","year":2025},"citing_paper":{"arxiv_id":"2605.22568","last_updated":"2026-05-21T14:47:54Z","snapshot_observed_at":"2026-08-16T05:58:07.197984Z","submitted_at":"2026-05-21T14:47:54Z","title":"Measuring Security Without Fooling Ourselves: Why Benchmarking Agents Is Hard","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-05-22T05:02:31.995894Z"},"links":{"cited_paper":"/paper/2507.02825","citing_paper":"/paper/2605.22568"},"observation_digest":"sha256:1f3761cd563ebb04ba56f87718a7e358ae5077edee131b35414c4884cd85302b","observation_id":"222d5930-73f8-4e41-83f8-2218daa2ec88","resolution":{"observed_at":"2026-05-22T05:04:37.290934Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.02825","last_updated":"2025-08-07T06:58:08Z","snapshot_observed_at":"2026-08-17T22:49:53.742436Z","submitted_at":"2025-07-03T17:35:31Z","title":"Establishing Best Practices for Building Rigorous Agentic Benchmarks","version":5},"cited_work":{"arxiv_id":"2507.02825","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2507.02825","snapshot_observed_at":"2026-07-03T22:08:59.930877Z","title":"Establishing best practices for building rigorous agentic benchmarks","venue":null,"work_id":"981c0760-99e5-4175-bca8-6be3ed159f5b","year":2025},"citing_paper":{"arxiv_id":"2605.22643","last_updated":"2026-05-22T14:53:30Z","snapshot_observed_at":"2026-08-16T12:49:45.643383Z","submitted_at":"2026-05-21T15:50:18Z","title":"Boiling the Frog: A Multi-Turn Benchmark for Agentic Safety","version":1},"reference_index":101,"source":"pdf_text","source_observed_at":"2026-05-22T05:50:28.114140Z"},"links":{"cited_paper":"/paper/2507.02825","citing_paper":"/paper/2605.22643"},"observation_digest":"sha256:c3524292ea8192564e85d539035c34f96852379d47d13b71c2feaa758d2ed600","observation_id":"fea22811-59c0-4626-9166-913c7c9b35e2","resolution":{"observed_at":"2026-05-22T05:51:07.934723Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.02825","last_updated":"2025-08-07T06:58:08Z","snapshot_observed_at":"2026-08-17T22:49:53.742436Z","submitted_at":"2025-07-03T17:35:31Z","title":"Establishing Best Practices for Building Rigorous Agentic Benchmarks","version":5},"cited_work":{"arxiv_id":"2507.02825","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2507.02825","snapshot_observed_at":"2026-07-03T22:08:59.930877Z","title":"Establishing best practices for building rigorous agentic benchmarks","venue":null,"work_id":"981c0760-99e5-4175-bca8-6be3ed159f5b","year":2025},"citing_paper":{"arxiv_id":"2605.22643","last_updated":"2026-05-22T14:53:30Z","snapshot_observed_at":"2026-08-16T12:49:45.643383Z","submitted_at":"2026-05-21T15:50:18Z","title":"Boiling the Frog: A Multi-Turn Benchmark for Agentic Safety","version":2},"reference_index":101,"source":"pdf_text","source_observed_at":"2026-05-25T06:05:27.736494Z"},"links":{"cited_paper":"/paper/2507.02825","citing_paper":"/paper/2605.22643"},"observation_digest":"sha256:6f0471e92ca1989270daf4cbcab87ab870baeec4ca73012ebcfff91b50911b15","observation_id":"a6c31d99-04db-4f81-9ad6-23df445ec369","resolution":{"observed_at":"2026-05-25T06:06:43.120173Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.02825","last_updated":"2025-08-07T06:58:08Z","snapshot_observed_at":"2026-08-17T22:49:53.742436Z","submitted_at":"2025-07-03T17:35:31Z","title":"Establishing Best Practices for Building Rigorous Agentic Benchmarks","version":5},"cited_work":{"arxiv_id":"2507.02825","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2507.02825","snapshot_observed_at":"2026-07-03T22:08:59.930877Z","title":"Establishing best practices for building rigorous agentic benchmarks","venue":null,"work_id":"981c0760-99e5-4175-bca8-6be3ed159f5b","year":2025},"citing_paper":{"arxiv_id":"2605.26195","last_updated":"2026-06-16T16:19:26Z","snapshot_observed_at":"2026-08-02T06:03:12.995535Z","submitted_at":"2026-05-25T16:26:59Z","title":"CyberEvolver: Structured Self-Evolution for Cybersecurity Agents On the Fly","version":2},"reference_index":80,"source":"pdf_text","source_observed_at":"2026-06-29T21:19:59.005348Z"},"links":{"cited_paper":"/paper/2507.02825","citing_paper":"/paper/2605.26195"},"observation_digest":"sha256:0677e013e91895ba25ba8d3384fe4854e157e9512b6d861c36cf7f518a63e037","observation_id":"5ce31bad-c793-4d0b-9344-a1c96088f20e","resolution":{"observed_at":"2026-06-29T21:23:58.970057Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.02825","last_updated":"2025-08-07T06:58:08Z","snapshot_observed_at":"2026-08-17T22:49:53.742436Z","submitted_at":"2025-07-03T17:35:31Z","title":"Establishing Best Practices for Building Rigorous Agentic Benchmarks","version":5},"cited_work":{"arxiv_id":"2507.02825","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2507.02825","snapshot_observed_at":"2026-07-03T22:08:59.930877Z","title":"Establishing best practices for building rigorous agentic benchmarks","venue":null,"work_id":"981c0760-99e5-4175-bca8-6be3ed159f5b","year":2025},"citing_paper":{"arxiv_id":"2605.26321","last_updated":"2026-05-25T20:44:17Z","snapshot_observed_at":"2026-08-14T12:08:17.496602Z","submitted_at":"2026-05-25T20:44:17Z","title":"Anchor: Mitigating Artifact Drift in Agent Benchmark Generation","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-06-29T21:29:57.626069Z"},"links":{"cited_paper":"/paper/2507.02825","citing_paper":"/paper/2605.26321"},"observation_digest":"sha256:e672258d004e6c2532247deb69bc44fb27c8b6817698d7072c528158800e44d6","observation_id":"f77ef2c1-260a-4d68-925f-e5722106874f","resolution":{"observed_at":"2026-06-29T21:33:59.135091Z","resolver_source":"arxiv_id","status":"malformed_identifier"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.02825","last_updated":"2025-08-07T06:58:08Z","snapshot_observed_at":"2026-08-17T22:49:53.742436Z","submitted_at":"2025-07-03T17:35:31Z","title":"Establishing Best Practices for Building Rigorous Agentic Benchmarks","version":5},"cited_work":{"arxiv_id":"2507.02825","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2507.02825","snapshot_observed_at":"2026-07-03T22:08:59.930877Z","title":"Establishing best practices for building rigorous agentic benchmarks","venue":null,"work_id":"981c0760-99e5-4175-bca8-6be3ed159f5b","year":2025},"citing_paper":{"arxiv_id":"2606.05391","last_updated":"2026-06-03T19:53:31Z","snapshot_observed_at":"2026-08-13T09:52:35.213263Z","submitted_at":"2026-06-03T19:53:31Z","title":"Human oversight of agentic systems in practice: Examining the oversight work, challenges, and heuristics of developers using software agents","version":1},"reference_index":157,"source":"pdf_text","source_observed_at":"2026-06-28T04:58:10.803420Z"},"links":{"cited_paper":"/paper/2507.02825","citing_paper":"/paper/2606.05391"},"observation_digest":"sha256:620fba46a38aebffdbf462f4f1576c0f348d87d14c38197f837bc2cf8a38d02b","observation_id":"366d0cce-affa-4028-baef-5e83e2554545","resolution":{"observed_at":"2026-07-02T10:36:52.845154Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.02825","last_updated":"2025-08-07T06:58:08Z","snapshot_observed_at":"2026-08-17T22:49:53.742436Z","submitted_at":"2025-07-03T17:35:31Z","title":"Establishing Best Practices for Building Rigorous Agentic Benchmarks","version":5},"cited_work":{"arxiv_id":"2507.02825","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2507.02825","snapshot_observed_at":"2026-07-03T22:08:59.930877Z","title":"Establishing best practices for building rigorous agentic benchmarks","venue":null,"work_id":"981c0760-99e5-4175-bca8-6be3ed159f5b","year":2025},"citing_paper":{"arxiv_id":"2606.18532","last_updated":"2026-06-16T22:57:24Z","snapshot_observed_at":"2026-08-17T01:13:56.343799Z","submitted_at":"2026-06-16T22:57:24Z","title":"AI Sandboxes: A Threat Model, Taxonomy, and Measurement Framework","version":1},"reference_index":167,"source":"pdf_text","source_observed_at":"2026-06-26T23:42:20.304205Z"},"links":{"cited_paper":"/paper/2507.02825","citing_paper":"/paper/2606.18532"},"observation_digest":"sha256:156f3438e11d1282583d037eff694639f21ef8a8b28173045b4222ed7b4e3be6","observation_id":"9390dd14-4f0c-450d-b790-a0e734359b67","resolution":{"observed_at":"2026-07-03T22:08:59.932826Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2507.02825","last_updated":"2025-08-07T06:58:08Z","snapshot_observed_at":"2026-08-17T22:49:53.742436Z","submitted_at":"2025-07-03T17:35:31Z","title":"Establishing Best Practices for Building Rigorous Agentic Benchmarks","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2507.02825","snapshot_observed_at":"2026-07-13T00:59:52.952364Z","title":"type-first","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.09016","last_updated":"2026-07-10T00:48:30Z","snapshot_observed_at":"2026-08-16T00:57:42.849342Z","submitted_at":"2026-07-10T00:48:30Z","title":"SLBench: Evaluating How LLM Agents Follow Logical Relations in Skills","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-07-13T00:59:52.952364Z"},"links":{"cited_paper":"/paper/2507.02825","citing_paper":"/paper/2607.09016"},"observation_digest":"sha256:89945c01154a9ba76d664209faecfd4a344d3e0b9fe394bc4316700ccb4b3c89","observation_id":"99f7b5c4-9deb-474e-9b48-e3d3354158c1","resolution":{"observed_at":"2026-07-13T00:59:52.952364Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2507.02825","last_updated":"2025-08-07T06:58:08Z","snapshot_observed_at":"2026-08-17T22:49:53.742436Z","submitted_at":"2025-07-03T17:35:31Z","title":"Establishing Best Practices for Building Rigorous Agentic Benchmarks","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2507.02825","snapshot_observed_at":"2026-08-01T23:34:10.122008Z","title":"Self-correction blind spot","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.15388","last_updated":"2026-07-16T18:38:10Z","snapshot_observed_at":"2026-08-20T15:06:39.321609Z","submitted_at":"2026-07-16T18:38:10Z","title":"Precise but Uncoupled: Reviewer Precision Does Not Guarantee Critique Uptake in Multi-Agent Math Reasoning","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-01T23:34:10.122008Z"},"links":{"cited_paper":"/paper/2507.02825","citing_paper":"/paper/2607.15388"},"observation_digest":"sha256:4d6a247d90a616f0674df419757e9fb9bfb8e65d18fd085dbaa9acd1ab9ffde7","observation_id":"ad9430a8-1636-4380-a69e-32bdca129aca","resolution":{"observed_at":"2026-08-01T23:34:10.122008Z","resolver_source":null,"status":"malformed_identifier"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2507.02825","last_updated":"2025-08-07T06:58:08Z","snapshot_observed_at":"2026-08-17T22:49:53.742436Z","submitted_at":"2025-07-03T17:35:31Z","title":"Establishing Best Practices for Building Rigorous Agentic Benchmarks","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2507.02825","snapshot_observed_at":"2026-08-02T10:00:38.648918Z","title":"Sekhon and Jacob Steinhardt and Antony Kellerman and Sarah Schwettmann and Matei Zaharia and Ion Stoica and Percy Liang and Daniel Kang , title =","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.16241","last_updated":"2026-06-26T03:22:50Z","snapshot_observed_at":"2026-08-18T06:32:01.627231Z","submitted_at":"2026-06-26T03:22:50Z","title":"KernelBench-Verified: Do LLM-Generated Kernels Actually Beat PyTorch?","version":1},"reference_index":24,"source":"arxiv_source","source_observed_at":"2026-08-02T10:00:38.648918Z"},"links":{"cited_paper":"/paper/2507.02825","citing_paper":"/paper/2607.16241"},"observation_digest":"sha256:f99cd10b955240605370776660f97e080678db7cca1dc23cd97689eac538a9aa","observation_id":"df467229-2c23-4b23-8ba3-c1307cd628b6","resolution":{"observed_at":"2026-08-02T10:00:38.648918Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2507.02825","last_updated":"2025-08-07T06:58:08Z","snapshot_observed_at":"2026-08-17T22:49:53.742436Z","submitted_at":"2025-07-03T17:35:31Z","title":"Establishing Best Practices for Building Rigorous Agentic Benchmarks","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2507.02825","snapshot_observed_at":"2026-08-01T15:04:09.728336Z","title":"Establishing best practices for building rigorous agentic benchmarks,","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.18575","last_updated":"2026-07-20T23:16:16Z","snapshot_observed_at":"2026-08-15T11:03:57.449469Z","submitted_at":"2026-07-20T23:16:16Z","title":"RECEIPT: Deterministic, Reward-Hacking-Resistant Verification for White-Box Agentic XSS Discovery","version":1},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-08-01T15:04:09.728336Z"},"links":{"cited_paper":"/paper/2507.02825","citing_paper":"/paper/2607.18575"},"observation_digest":"sha256:2d5575285f84108ba4b8dc38aad87c9b508cd1052c54a28576889cc90707788a","observation_id":"4f0e28b2-9010-4193-a5c6-d8e3470bbf1a","resolution":{"observed_at":"2026-08-01T15:04:09.728336Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2507.02825","last_updated":"2025-08-07T06:58:08Z","snapshot_observed_at":"2026-08-17T22:49:53.742436Z","submitted_at":"2025-07-03T17:35:31Z","title":"Establishing Best Practices for Building Rigorous Agentic Benchmarks","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2507.02825","snapshot_observed_at":"2026-08-01T12:54:38.319817Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.19292","last_updated":"2026-07-21T17:02:37Z","snapshot_observed_at":"2026-08-14T07:56:28.626682Z","submitted_at":"2026-07-21T17:02:37Z","title":"The safety failures we are not instrumenting: a perspective on hidden safety-critical challenges in modern AI systems","version":1},"reference_index":71,"source":"pdf_text","source_observed_at":"2026-08-01T12:54:38.319817Z"},"links":{"cited_paper":"/paper/2507.02825","citing_paper":"/paper/2607.19292"},"observation_digest":"sha256:b51c57ba670e13a745c177cc1950e1bab223606c5987149d5eaaa9691d600844","observation_id":"e573652e-ca9f-4e82-a458-090e7836e591","resolution":{"observed_at":"2026-08-01T12:54:38.319817Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2507.02825","last_updated":"2025-08-07T06:58:08Z","snapshot_observed_at":"2026-08-17T22:49:53.742436Z","submitted_at":"2025-07-03T17:35:31Z","title":"Establishing Best Practices for Building Rigorous Agentic Benchmarks","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2507.02825","snapshot_observed_at":"2026-08-01T06:49:28.621324Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.21763","last_updated":"2026-07-23T19:26:25Z","snapshot_observed_at":"2026-08-14T16:17:34.986609Z","submitted_at":"2026-07-23T19:26:25Z","title":"Every Model Cheats: Prompt-Level Mitigation of Cheating on Offensive Cyber Tasks","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-01T06:49:28.621324Z"},"links":{"cited_paper":"/paper/2507.02825","citing_paper":"/paper/2607.21763"},"observation_digest":"sha256:e60beb15ef192e479d2bb899aaa5137dddaf775b99ce1a3b5da645d66c282eab","observation_id":"6f262067-0274-4d99-92e7-06ea467ed41a","resolution":{"observed_at":"2026-08-01T06:49:28.621324Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2507.02825","last_updated":"2025-08-07T06:58:08Z","snapshot_observed_at":"2026-08-17T22:49:53.742436Z","submitted_at":"2025-07-03T17:35:31Z","title":"Establishing Best Practices for Building Rigorous Agentic Benchmarks","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2507.02825","snapshot_observed_at":"2026-07-31T23:58:27.966370Z","title":"arXiv preprint arXiv:2507.02825 , year =","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.23263","last_updated":"2026-07-25T16:00:45Z","snapshot_observed_at":"2026-08-19T23:09:28.604854Z","submitted_at":"2026-07-25T16:00:45Z","title":"SeekJudge: A Practical Reward Framework for Reinforcement Learning in Computer-Use Agents","version":1},"reference_index":12,"source":"arxiv_source","source_observed_at":"2026-07-31T23:58:27.966370Z"},"links":{"cited_paper":"/paper/2507.02825","citing_paper":"/paper/2607.23263"},"observation_digest":"sha256:11e0a43c3fc4f8723ae633c739d99c6eea0cbedb69f1273647c539dd7ca30dad","observation_id":"65fb692f-d7ab-4f20-8921-75e04f349ef3","resolution":{"observed_at":"2026-07-31T23:58:27.966370Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2507.02825","last_updated":"2025-08-07T06:58:08Z","snapshot_observed_at":"2026-08-17T22:49:53.742436Z","submitted_at":"2025-07-03T17:35:31Z","title":"Establishing Best Practices for Building Rigorous Agentic Benchmarks","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2507.02825","snapshot_observed_at":"2026-08-01T06:24:59.912539Z","title":"Content-Type: application/json","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.27518","last_updated":"2026-07-29T23:17:08Z","snapshot_observed_at":"2026-08-14T10:56:41.548668Z","submitted_at":"2026-07-29T23:17:08Z","title":"Automated Transcript Analysis for Detecting Flaws in Agentic Benchmarks","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-01T06:24:59.912539Z"},"links":{"cited_paper":"/paper/2507.02825","citing_paper":"/paper/2607.27518"},"observation_digest":"sha256:375fb7fe37eb21f909e21883093a2f4549d70dba1df17716461981095c286a67","observation_id":"df392865-68cb-4dc7-9285-56fc59986a67","resolution":{"observed_at":"2026-08-01T06:24:59.912539Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2507.02825","last_updated":"2025-08-07T06:58:08Z","snapshot_observed_at":"2026-08-17T22:49:53.742436Z","submitted_at":"2025-07-03T17:35:31Z","title":"Establishing Best Practices for Building Rigorous Agentic Benchmarks","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2507.02825","snapshot_observed_at":"2026-08-03T00:25:08.204255Z","title":"arXiv preprint arXiv:2507.02825 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2607.28802","last_updated":"2026-07-30T19:55:14Z","snapshot_observed_at":"2026-08-08T06:19:07.682563Z","submitted_at":"2026-07-30T19:55:14Z","title":"Model or Harness? An Interaction-Centric Taxonomy for Localizing Agent Failures","version":1},"reference_index":26,"source":"arxiv_source","source_observed_at":"2026-08-03T00:25:08.204255Z"},"links":{"cited_paper":"/paper/2507.02825","citing_paper":"/paper/2607.28802"},"observation_digest":"sha256:bec8754ba92a2fcae374daa49a8964609a87e72858540f8eb590b877db86b476","observation_id":"897a0c3a-b537-43ee-b4ea-d7abc981c559","resolution":{"observed_at":"2026-08-03T00:25:08.204255Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2507.02825","last_updated":"2025-08-07T06:58:08Z","snapshot_observed_at":"2026-08-17T22:49:53.742436Z","submitted_at":"2025-07-03T17:35:31Z","title":"Establishing Best Practices for Building Rigorous Agentic Benchmarks","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2507.02825","snapshot_observed_at":"2026-08-04T00:55:48.271679Z","title":"arXiv preprint arXiv:2507.02825 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.00267","last_updated":"2026-08-10T12:29:11Z","snapshot_observed_at":"2026-08-16T07:45:36.109671Z","submitted_at":"2026-07-31T20:18:25Z","title":"LoopsBench: From Harness Engineering to Loop Engineering in Coding Agent Evaluation","version":1},"reference_index":36,"source":"arxiv_source","source_observed_at":"2026-08-04T00:55:48.271679Z"},"links":{"cited_paper":"/paper/2507.02825","citing_paper":"/paper/2608.00267"},"observation_digest":"sha256:e898bc45f1530d2877ea320854bab172cc2ecb51866ca902c40bdc9ffac2f54e","observation_id":"1d6a3c2d-ea8a-4eed-8871-45486bfd7d0f","resolution":{"observed_at":"2026-08-04T00:55:48.271679Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2507.02825","last_updated":"2025-08-07T06:58:08Z","snapshot_observed_at":"2026-08-17T22:49:53.742436Z","submitted_at":"2025-07-03T17:35:31Z","title":"Establishing Best Practices for Building Rigorous Agentic Benchmarks","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2507.02825","snapshot_observed_at":"2026-08-05T04:18:50.856828Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.00794","last_updated":"2026-08-05T05:37:25Z","snapshot_observed_at":"2026-08-13T15:26:52.647655Z","submitted_at":"2026-08-01T17:50:12Z","title":"Measurement Without Validity: The Compounding Reliability Problem in Agentic AI Evaluation","version":2},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-05T04:18:50.856828Z"},"links":{"cited_paper":"/paper/2507.02825","citing_paper":"/paper/2608.00794"},"observation_digest":"sha256:9edb3e4669246f9139b029680f3dcb965608f856507c9a5d9f4e2bf63d5b1c40","observation_id":"7738450d-3702-4acb-a2eb-5776ee593f3e","resolution":{"observed_at":"2026-08-05T04:18:50.856828Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2507.02825","last_updated":"2025-08-07T06:58:08Z","snapshot_observed_at":"2026-08-17T22:49:53.742436Z","submitted_at":"2025-07-03T17:35:31Z","title":"Establishing Best Practices for Building Rigorous Agentic Benchmarks","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2507.02825","snapshot_observed_at":"2026-08-06T04:16:24.726546Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.00794","last_updated":"2026-08-05T05:37:25Z","snapshot_observed_at":"2026-08-13T15:26:52.647655Z","submitted_at":"2026-08-01T17:50:12Z","title":"Measurement Without Validity: The Compounding Reliability Problem in Agentic AI Evaluation","version":3},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-06T04:16:24.726546Z"},"links":{"cited_paper":"/paper/2507.02825","citing_paper":"/paper/2608.00794"},"observation_digest":"sha256:b9dcbc6808d07144b8afd3298939f001a20681258909350e2172a97ba3b166ad","observation_id":"a2274c8f-1274-4080-9b81-a586d4d723cb","resolution":{"observed_at":"2026-08-06T04:16:24.726546Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2507.02825","last_updated":"2025-08-07T06:58:08Z","snapshot_observed_at":"2026-08-17T22:49:53.742436Z","submitted_at":"2025-07-03T17:35:31Z","title":"Establishing Best Practices for Building Rigorous Agentic Benchmarks","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2507.02825","snapshot_observed_at":"2026-08-10T23:01:35.480747Z","title":"Establishing best practices for building rigorous agentic benchmarks, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.06663","last_updated":"2026-08-07T00:19:48Z","snapshot_observed_at":"2026-08-20T07:06:19.804536Z","submitted_at":"2026-08-07T00:19:48Z","title":"The Horizon Gap: Planning, Memory, Execution, Training, and Evaluation for Long-Horizon LLM Agents","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-10T23:01:35.480747Z"},"links":{"cited_paper":"/paper/2507.02825","citing_paper":"/paper/2608.06663"},"observation_digest":"sha256:1bc5fe09b16fdbee904eca071fde446b7f252daed5bc1b04d695c467d80e9760","observation_id":"0cf16eda-132d-4502-aa08-c8cad43c3488","resolution":{"observed_at":"2026-08-10T23:01:35.480747Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2507.02825","last_updated":"2025-08-07T06:58:08Z","snapshot_observed_at":"2026-08-17T22:49:53.742436Z","submitted_at":"2025-07-03T17:35:31Z","title":"Establishing Best Practices for Building Rigorous Agentic Benchmarks","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2507.02825","snapshot_observed_at":"2026-08-15T14:19:14.004821Z","title":null,"venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2608.11274","last_updated":"2026-08-11T08:01:05Z","snapshot_observed_at":"2026-08-19T03:07:49.206917Z","submitted_at":"2026-08-11T08:01:05Z","title":"Agent Safety Should Be a Runtime Contract","version":1},"reference_index":100,"source":"pdf_text","source_observed_at":"2026-08-15T14:19:14.004821Z"},"links":{"cited_paper":"/paper/2507.02825","citing_paper":"/paper/2608.11274"},"observation_digest":"sha256:4153fb890e90582dc3c724687ca7cdbd9db4e0a52812c7eb2eb7eb1d35a5d29e","observation_id":"e2b186c1-b8e7-4440-af39-d8191b3b9363","resolution":{"observed_at":"2026-08-15T14:19:14.004821Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2507.02825/citation-record","integrity":"/paper/2507.02825/integrity","json":"/paper/2507.02825/citation-record.json","paper":"/paper/2507.02825"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:24:19.476468Z","title":"Inspect AI: Framework for Large Language Model Evaluations, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.02825","last_updated":"2025-08-07T06:58:08Z","snapshot_observed_at":"2026-08-17T22:49:53.742436Z","submitted_at":"2025-07-03T17:35:31Z","title":"Establishing Best Practices for Building Rigorous Agentic Benchmarks","version":5},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-08-06T20:24:19.476468Z"},"links":{"citing_paper":"/paper/2507.02825"},"observation_digest":"sha256:c1c8b767ce01d607cd259ed7cede10d7ed08b5625df9ed120e6332381de82527","observation_id":"498da9e5-9da9-4859-a1f0-e2235cb99aaf","resolution":{"observed_at":"2026-08-06T20:24:19.476468Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:24:19.643317Z","title":"Gpt code editing benchmarks, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.02825","last_updated":"2025-08-07T06:58:08Z","snapshot_observed_at":"2026-08-17T22:49:53.742436Z","submitted_at":"2025-07-03T17:35:31Z","title":"Establishing Best Practices for Building Rigorous Agentic Benchmarks","version":5},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-06T20:24:19.643317Z"},"links":{"citing_paper":"/paper/2507.02825"},"observation_digest":"sha256:a3939d90e62abc382939c5c0f02c96b8f1a116cc0c40e991832eb69a86e64b63","observation_id":"5aea9622-5a6f-4b7e-87c6-96ad04a561b7","resolution":{"observed_at":"2026-08-06T20:24:19.643317Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:24:19.860736Z","title":"o1 tops aider’s new polyglot leaderboard, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.02825","last_updated":"2025-08-07T06:58:08Z","snapshot_observed_at":"2026-08-17T22:49:53.742436Z","submitted_at":"2025-07-03T17:35:31Z","title":"Establishing Best Practices for Building Rigorous Agentic Benchmarks","version":5},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-08-06T20:24:19.860736Z"},"links":{"citing_paper":"/paper/2507.02825"},"observation_digest":"sha256:f34e0196266ac72707b1e2d516fb0d50c9241bd1216eec2b60e52132deda7ac4","observation_id":"e9a79027-ed54-47d7-a5ac-f01f5ccc7ccb","resolution":{"observed_at":"2026-08-06T20:24:19.860736Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:24:20.062843Z","title":"The amazon nova family of models: Technical report and model card, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.02825","last_updated":"2025-08-07T06:58:08Z","snapshot_observed_at":"2026-08-17T22:49:53.742436Z","submitted_at":"2025-07-03T17:35:31Z","title":"Establishing Best Practices for Building Rigorous Agentic Benchmarks","version":5},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-06T20:24:20.062843Z"},"links":{"citing_paper":"/paper/2507.02825"},"observation_digest":"sha256:5c8b35684357249ac0f973ec56abd59842859c2a1a8b83938b8b1a34dd778538","observation_id":"0b486f1a-5d0e-437d-962b-35608d837b1b","resolution":{"observed_at":"2026-08-06T20:24:20.062843Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:24:20.190930Z","title":"Claude 3.5 sonnet, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.02825","last_updated":"2025-08-07T06:58:08Z","snapshot_observed_at":"2026-08-17T22:49:53.742436Z","submitted_at":"2025-07-03T17:35:31Z","title":"Establishing Best Practices for Building Rigorous Agentic Benchmarks","version":5},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-06T20:24:20.190930Z"},"links":{"citing_paper":"/paper/2507.02825"},"observation_digest":"sha256:b81ca09b2cf05fb25ab3bdc2fb4834748f2ea7763b3db1c729f604df61bcaae5","observation_id":"1d636119-8fda-4571-83bd-dc153e8716cd","resolution":{"observed_at":"2026-08-06T20:24:20.190930Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:24:20.342592Z","title":"Claude 3.7 and claude code, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.02825","last_updated":"2025-08-07T06:58:08Z","snapshot_observed_at":"2026-08-17T22:49:53.742436Z","submitted_at":"2025-07-03T17:35:31Z","title":"Establishing Best Practices for Building Rigorous Agentic Benchmarks","version":5},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-06T20:24:20.342592Z"},"links":{"citing_paper":"/paper/2507.02825"},"observation_digest":"sha256:e4e035b34637dbfaea09d769b43fa6afb2f55d0bedbc9d0b650cf9836be311aa","observation_id":"1623b2cf-dbaa-48cb-80e9-ba7df2364368","resolution":{"observed_at":"2026-08-06T20:24:20.342592Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:24:20.521292Z","title":"Bird minidev - corrections, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.02825","last_updated":"2025-08-07T06:58:08Z","snapshot_observed_at":"2026-08-17T22:49:53.742436Z","submitted_at":"2025-07-03T17:35:31Z","title":"Establishing Best Practices for Building Rigorous Agentic Benchmarks","version":5},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-06T20:24:20.521292Z"},"links":{"citing_paper":"/paper/2507.02825"},"observation_digest":"sha256:311c17120a8ee72f11fc4abe922f51038fddfb7b0d966ebfda88065d4a06aa84","observation_id":"d89439eb-2c9e-43cd-86d2-2223c0d8a01d","resolution":{"observed_at":"2026-08-06T20:24:20.521292Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2108.07732","last_updated":"2021-08-16T03:57:30Z","snapshot_observed_at":"2026-08-15T17:40:38.050939Z","submitted_at":"2021-08-16T03:57:30Z","title":"Program Synthesis with Large Language Models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2108.07732","snapshot_observed_at":"2026-08-06T20:24:20.658359Z","title":"Program synthesis with large language models","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2507.02825","last_updated":"2025-08-07T06:58:08Z","snapshot_observed_at":"2026-08-17T22:49:53.742436Z","submitted_at":"2025-07-03T17:35:31Z","title":"Establishing Best Practices for Building Rigorous Agentic Benchmarks","version":5},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-06T20:24:20.658359Z"},"links":{"cited_paper":"/paper/2108.07732","citing_paper":"/paper/2507.02825"},"observation_digest":"sha256:fd02a79873d6743b59146c8259cfb72991d8cc468ed301d556beeecc3f65083a","observation_id":"59767821-da00-4625-abeb-e8755f7f14cf","resolution":{"observed_at":"2026-08-06T20:24:20.658359Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.18403","last_updated":"2025-06-02T11:31:19Z","snapshot_observed_at":"2026-08-16T14:19:08.369776Z","submitted_at":"2024-06-26T14:56:13Z","title":"LLMs instead of Human Judges? A Large Scale Empirical Study across 20 NLP Evaluation Tasks","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.18403","snapshot_observed_at":"2026-08-06T20:24:20.815544Z","title":"Llms instead of human judges? a large scale empirical study across 20 nlp evaluation tasks","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.02825","last_updated":"2025-08-07T06:58:08Z","snapshot_observed_at":"2026-08-17T22:49:53.742436Z","submitted_at":"2025-07-03T17:35:31Z","title":"Establishing Best Practices for Building Rigorous Agentic Benchmarks","version":5},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-06T20:24:20.815544Z"},"links":{"cited_paper":"/paper/2406.18403","citing_paper":"/paper/2507.02825"},"observation_digest":"sha256:5de8c9436282004c9b4d2e498f15808d12166e3749ed894368ff3fa48824a62e","observation_id":"eb83d160-535e-4026-9385-50edf7de0464","resolution":{"observed_at":"2026-08-06T20:24:20.815544Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.10711","last_updated":"2026-07-06T06:02:32Z","snapshot_observed_at":"2026-08-20T02:10:29.773827Z","submitted_at":"2025-01-18T09:51:57Z","title":"Code Benchmarks Should Prioritize Rigor, Reliability, and Reproducibility","version":5},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.10711","snapshot_observed_at":"2026-08-06T20:24:20.949000Z","title":"How should i build a benchmark? arXiv preprint arXiv:2501.10711, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.02825","last_updated":"2025-08-07T06:58:08Z","snapshot_observed_at":"2026-08-17T22:49:53.742436Z","submitted_at":"2025-07-03T17:35:31Z","title":"Establishing Best Practices for Building Rigorous Agentic Benchmarks","version":5},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-06T20:24:20.949000Z"},"links":{"cited_paper":"/paper/2501.10711","citing_paper":"/paper/2507.02825"},"observation_digest":"sha256:3c44217b6554b008a8748e217b1d446d6d49841d864ec166010d271fc86b9788","observation_id":"58917dc5-a8db-4654-8d3a-2a65a00bfc8e","resolution":{"observed_at":"2026-08-06T20:24:20.949000Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.07095","last_updated":"2025-02-26T11:57:30Z","snapshot_observed_at":"2026-08-12T16:49:10.970507Z","submitted_at":"2024-10-09T17:34:27Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.07095","snapshot_observed_at":"2026-08-06T20:24:21.069258Z","title":"Mle-bench: Evaluating machine learning agents on machine learning engineering","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.02825","last_updated":"2025-08-07T06:58:08Z","snapshot_observed_at":"2026-08-17T22:49:53.742436Z","submitted_at":"2025-07-03T17:35:31Z","title":"Establishing Best Practices for Building Rigorous Agentic Benchmarks","version":5},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-06T20:24:21.069258Z"},"links":{"cited_paper":"/paper/2410.07095","citing_paper":"/paper/2507.02825"},"observation_digest":"sha256:efebb8ffbaf4b9e3d9d652e1ec8aa0d876c41a8a72924d029c9320474e57bad7","observation_id":"c3080b94-bee8-43c4-a758-4a0b10a01473","resolution":{"observed_at":"2026-08-06T20:24:21.069258Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2309.17288","last_updated":"2024-04-29T18:38:26Z","snapshot_observed_at":"2026-08-16T14:56:29.843666Z","submitted_at":"2023-09-29T14:46:30Z","title":"AutoAgents: A Framework for Automatic Agent Generation","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2309.17288","snapshot_observed_at":"2026-08-06T20:24:21.195372Z","title":"Autoagents: A framework for automatic agent generation","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.02825","last_updated":"2025-08-07T06:58:08Z","snapshot_observed_at":"2026-08-17T22:49:53.742436Z","submitted_at":"2025-07-03T17:35:31Z","title":"Establishing Best Practices for Building Rigorous Agentic Benchmarks","version":5},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-08-06T20:24:21.195372Z"},"links":{"cited_paper":"/paper/2309.17288","citing_paper":"/paper/2507.02825"},"observation_digest":"sha256:92c8677d9596fbf8772023e9abf8dc5a675d23f741578c63c2db6c287d53f8a9","observation_id":"260b815f-fb0f-4913-ab43-8dc7a875ea78","resolution":{"observed_at":"2026-08-06T20:24:21.195372Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:24:21.316891Z","title":null,"venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2507.02825","last_updated":"2025-08-07T06:58:08Z","snapshot_observed_at":"2026-08-17T22:49:53.742436Z","submitted_at":"2025-07-03T17:35:31Z","title":"Establishing Best Practices for Building Rigorous Agentic Benchmarks","version":5},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-08-06T20:24:21.316891Z"},"links":{"citing_paper":"/paper/2507.02825"},"observation_digest":"sha256:ed93842fc146f8ea8ae323eac3e8d85bcceab12d77de7f3ed2cb66d082ac0149","observation_id":"5bccefc2-874b-4a7b-9091-43df7c83c51c","resolution":{"observed_at":"2026-08-06T20:24:21.316891Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:24:21.435890Z","title":"Jimenez, John Yang, Leyton Ho, Tejal Patwardhan, Kevin Liu, and Aleksander Madry","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.02825","last_updated":"2025-08-07T06:58:08Z","snapshot_observed_at":"2026-08-17T22:49:53.742436Z","submitted_at":"2025-07-03T17:35:31Z","title":"Establishing Best Practices for Building Rigorous Agentic Benchmarks","version":5},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-06T20:24:21.435890Z"},"links":{"citing_paper":"/paper/2507.02825"},"observation_digest":"sha256:37df6cc0c09835635cd0fcd0564bc4d9b5f5dfb7f0192c803d530022d9c9d63a","observation_id":"78deee31-97dc-42af-af85-ad8058691468","resolution":{"observed_at":"2026-08-06T20:24:21.435890Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:24:21.523208Z","title":"Introducing deepseek v3, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.02825","last_updated":"2025-08-07T06:58:08Z","snapshot_observed_at":"2026-08-17T22:49:53.742436Z","submitted_at":"2025-07-03T17:35:31Z","title":"Establishing Best Practices for Building Rigorous Agentic Benchmarks","version":5},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-06T20:24:21.523208Z"},"links":{"citing_paper":"/paper/2507.02825"},"observation_digest":"sha256:f6342872255208770776f8058e27bdef9fc1127df20471c28ff2f22729ca57e4","observation_id":"0a80de66-d368-488c-ac16-16624a5bda46","resolution":{"observed_at":"2026-08-06T20:24:21.523208Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:24:21.651866Z","title":"Imagenet: A large- scale hierarchical image database","venue":null,"work_id":null,"year":2009},"citing_paper":{"arxiv_id":"2507.02825","last_updated":"2025-08-07T06:58:08Z","snapshot_observed_at":"2026-08-17T22:49:53.742436Z","submitted_at":"2025-07-03T17:35:31Z","title":"Establishing Best Practices for Building Rigorous Agentic Benchmarks","version":5},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-06T20:24:21.651866Z"},"links":{"citing_paper":"/paper/2507.02825"},"observation_digest":"sha256:f2c0bcf5446c9db7e4096daef7e3a6d52da47344d805e826ecac287a3f2933c9","observation_id":"7c7a4f70-66f9-45fb-8a76-6496e199b5b6","resolution":{"observed_at":"2026-08-06T20:24:21.651866Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:24:21.733040Z","title":"Don’t label twice: Quantity beats quality when comparing binary classifiers on a budget","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.02825","last_updated":"2025-08-07T06:58:08Z","snapshot_observed_at":"2026-08-17T22:49:53.742436Z","submitted_at":"2025-07-03T17:35:31Z","title":"Establishing Best Practices for Building Rigorous Agentic Benchmarks","version":5},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-06T20:24:21.733040Z"},"links":{"citing_paper":"/paper/2507.02825"},"observation_digest":"sha256:b123afc82eadb5a6bbb97c257f0c036769bbaa4fa23f7bdcfcd22a0d8ec2ff74","observation_id":"3e3cf797-2521-4f5a-9c09-cc752903aa01","resolution":{"observed_at":"2026-08-06T20:24:21.733040Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:24:21.837718Z","title":"Limits to scalable evaluation at the frontier: Llm as judge won’t beat twice the data","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.02825","last_updated":"2025-08-07T06:58:08Z","snapshot_observed_at":"2026-08-17T22:49:53.742436Z","submitted_at":"2025-07-03T17:35:31Z","title":"Establishing Best Practices for Building Rigorous Agentic Benchmarks","version":5},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-06T20:24:21.837718Z"},"links":{"citing_paper":"/paper/2507.02825"},"observation_digest":"sha256:fc52920bd2673ea7e2e30c787b3933345f43459c16b432a96dacdffd985ea387","observation_id":"9f23249f-8e95-4619-8525-f6f7f568a9a5","resolution":{"observed_at":"2026-08-06T20:24:21.837718Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:24:21.946111Z","title":"The design and operation of CloudLab","venue":null,"work_id":null,"year":2019},"citing_paper":{"arxiv_id":"2507.02825","last_updated":"2025-08-07T06:58:08Z","snapshot_observed_at":"2026-08-17T22:49:53.742436Z","submitted_at":"2025-07-03T17:35:31Z","title":"Establishing Best Practices for Building Rigorous Agentic Benchmarks","version":5},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-06T20:24:21.946111Z"},"links":{"citing_paper":"/paper/2507.02825"},"observation_digest":"sha256:b7a1af75ec7080da55e935c68af16fc916ccf2e70504df67a363d1921b2638fe","observation_id":"03881fb7-d8f6-4f03-b3eb-8aae63ca2da1","resolution":{"observed_at":"2026-08-06T20:24:21.946111Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.06559","last_updated":"2025-05-25T23:36:47Z","snapshot_observed_at":"2026-08-19T05:02:22.089696Z","submitted_at":"2025-02-10T15:25:06Z","title":"Can We Trust AI Benchmarks? An Interdisciplinary Review of Current Issues in AI Evaluation","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.06559","snapshot_observed_at":"2026-08-06T20:24:22.007624Z","title":"Can we trust ai benchmarks? an interdisciplinary review of current issues in ai evaluation","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.02825","last_updated":"2025-08-07T06:58:08Z","snapshot_observed_at":"2026-08-17T22:49:53.742436Z","submitted_at":"2025-07-03T17:35:31Z","title":"Establishing Best Practices for Building Rigorous Agentic Benchmarks","version":5},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-08-06T20:24:22.007624Z"},"links":{"cited_paper":"/paper/2502.06559","citing_paper":"/paper/2507.02825"},"observation_digest":"sha256:d30c7e3475bc49e681e0c66ae7f4aba532dd8e5d80ee9a3bda85cdb875a55bc3","observation_id":"06be70c0-2a6c-4b6d-9947-589ce3f0d3ab","resolution":{"observed_at":"2026-08-06T20:24:22.007624Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:24:22.076812Z","title":"Searching for computer vision north stars","venue":null,"work_id":null,"year":2022},"citing_paper":{"arxiv_id":"2507.02825","last_updated":"2025-08-07T06:58:08Z","snapshot_observed_at":"2026-08-17T22:49:53.742436Z","submitted_at":"2025-07-03T17:35:31Z","title":"Establishing Best Practices for Building Rigorous Agentic Benchmarks","version":5},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-06T20:24:22.076812Z"},"links":{"citing_paper":"/paper/2507.02825"},"observation_digest":"sha256:1bf1367bba34f707585b8e5c78c43051b7737a9b28ccb34626dbcba6066a0471","observation_id":"b9413ed1-7af1-4651-a911-5f97e31cd44d","resolution":{"observed_at":"2026-08-06T20:24:22.076812Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.04872","last_updated":"2025-12-23T02:23:47Z","snapshot_observed_at":"2026-08-13T13:45:44.430193Z","submitted_at":"2024-11-07T17:07:35Z","title":"FrontierMath: A Benchmark for Evaluating Advanced Mathematical Reasoning in AI","version":7},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.04872","snapshot_observed_at":"2026-08-06T20:24:22.159924Z","title":"Frontiermath: A benchmark for evaluating advanced mathematical reasoning in ai","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.02825","last_updated":"2025-08-07T06:58:08Z","snapshot_observed_at":"2026-08-17T22:49:53.742436Z","submitted_at":"2025-07-03T17:35:31Z","title":"Establishing Best Practices for Building Rigorous Agentic Benchmarks","version":5},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-08-06T20:24:22.159924Z"},"links":{"cited_paper":"/paper/2411.04872","citing_paper":"/paper/2507.02825"},"observation_digest":"sha256:621c33a1ef266df3080f6a8f0d5bcb9192fef1b0c32628ee2ac8e8642a1980c0","observation_id":"2fc04ec4-6484-4b78-9075-bd038d42f644","resolution":{"observed_at":"2026-08-06T20:24:22.159924Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:24:22.247550Z","title":"A classification of sql injection attacks and countermeasures","venue":null,"work_id":null,"year":2006},"citing_paper":{"arxiv_id":"2507.02825","last_updated":"2025-08-07T06:58:08Z","snapshot_observed_at":"2026-08-17T22:49:53.742436Z","submitted_at":"2025-07-03T17:35:31Z","title":"Establishing Best Practices for Building Rigorous Agentic Benchmarks","version":5},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-08-06T20:24:22.247550Z"},"links":{"citing_paper":"/paper/2507.02825"},"observation_digest":"sha256:dbbc136abb5d2c20b281a19ecfba53e18e844a2b40980973f25f240869bfe769","observation_id":"9d379cf5-0b6c-4177-a756-93c27c27eeeb","resolution":{"observed_at":"2026-08-06T20:24:22.247550Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:24:22.337945Z","title":"More than marketing? on the information value of ai benchmarks for practitioners","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.02825","last_updated":"2025-08-07T06:58:08Z","snapshot_observed_at":"2026-08-17T22:49:53.742436Z","submitted_at":"2025-07-03T17:35:31Z","title":"Establishing Best Practices for Building Rigorous Agentic Benchmarks","version":5},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-06T20:24:22.337945Z"},"links":{"citing_paper":"/paper/2507.02825"},"observation_digest":"sha256:a7fd7ad003ddba9915fddb7b12a444ce0d0c2099249a574b07d5272297b825ab","observation_id":"4706c5b4-9b26-4c38-8cf0-7db4d811c7d3","resolution":{"observed_at":"2026-08-06T20:24:22.337945Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2401.13919","last_updated":"2024-06-06T18:37:34Z","snapshot_observed_at":"2026-08-15T01:23:25.524198Z","submitted_at":"2024-01-25T03:33:18Z","title":"WebVoyager: Building an End-to-End Web Agent with Large Multimodal Models","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2401.13919","snapshot_observed_at":"2026-08-06T20:24:22.424552Z","title":"Webvoyager: Building an end-to-end web agent with large multimodal models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.02825","last_updated":"2025-08-07T06:58:08Z","snapshot_observed_at":"2026-08-17T22:49:53.742436Z","submitted_at":"2025-07-03T17:35:31Z","title":"Establishing Best Practices for Building Rigorous Agentic Benchmarks","version":5},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-08-06T20:24:22.424552Z"},"links":{"cited_paper":"/paper/2401.13919","citing_paper":"/paper/2507.02825"},"observation_digest":"sha256:ac383f7ba61e4c04d69feae9b2c09aeb50b5559dc8acc87b5162722bf7af1bb8","observation_id":"8e102cf6-4a50-42e7-9ccf-abe7fefd4706","resolution":{"observed_at":"2026-08-06T20:24:22.424552Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:24:22.505161Z","title":"The extent and consequences of p-hacking in science","venue":null,"work_id":null,"year":2015},"citing_paper":{"arxiv_id":"2507.02825","last_updated":"2025-08-07T06:58:08Z","snapshot_observed_at":"2026-08-17T22:49:53.742436Z","submitted_at":"2025-07-03T17:35:31Z","title":"Establishing Best Practices for Building Rigorous Agentic Benchmarks","version":5},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-08-06T20:24:22.505161Z"},"links":{"citing_paper":"/paper/2507.02825"},"observation_digest":"sha256:3f769bba952bb1d75e1e2088dfc0f4abb3fcb147534dd64dadedacf0fa9ac837","observation_id":"fdc8141c-a335-4548-9956-ecba7fc35208","resolution":{"observed_at":"2026-08-06T20:24:22.505161Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2009.03300","last_updated":"2021-01-12T18:57:11Z","snapshot_observed_at":"2026-08-13T20:44:28.824685Z","submitted_at":"2020-09-07T17:59:25Z","title":"Measuring Massive Multitask Language Understanding","version":3},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2009.03300","snapshot_observed_at":"2026-08-06T20:24:22.595920Z","title":"Measuring massive multitask language understanding","venue":null,"work_id":null,"year":2009},"citing_paper":{"arxiv_id":"2507.02825","last_updated":"2025-08-07T06:58:08Z","snapshot_observed_at":"2026-08-17T22:49:53.742436Z","submitted_at":"2025-07-03T17:35:31Z","title":"Establishing Best Practices for Building Rigorous Agentic Benchmarks","version":5},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-08-06T20:24:22.595920Z"},"links":{"cited_paper":"/paper/2009.03300","citing_paper":"/paper/2507.02825"},"observation_digest":"sha256:04660a15364fae80c025945d28e53246e0a67b3addc72db4c22a1d1e17254ff9","observation_id":"36a63e39-97b0-4b90-995f-a6e0a43c8980","resolution":{"observed_at":"2026-08-06T20:24:22.595920Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:24:22.654448Z","title":"The design and analysis of benchmark experiments","venue":null,"work_id":null,"year":2005},"citing_paper":{"arxiv_id":"2507.02825","last_updated":"2025-08-07T06:58:08Z","snapshot_observed_at":"2026-08-17T22:49:53.742436Z","submitted_at":"2025-07-03T17:35:31Z","title":"Establishing Best Practices for Building Rigorous Agentic Benchmarks","version":5},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-08-06T20:24:22.654448Z"},"links":{"citing_paper":"/paper/2507.02825"},"observation_digest":"sha256:ceaba26e763a17603e246e8b9807b578b81a1b75a2fd83a7f3dafcd354f15e15","observation_id":"638e800b-6160-41d1-ba91-ddee087bb69f","resolution":{"observed_at":"2026-08-06T20:24:22.654448Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:24:22.744781Z","title":"Preventing server-side request forgery attacks","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2507.02825","last_updated":"2025-08-07T06:58:08Z","snapshot_observed_at":"2026-08-17T22:49:53.742436Z","submitted_at":"2025-07-03T17:35:31Z","title":"Establishing Best Practices for Building Rigorous Agentic Benchmarks","version":5},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-08-06T20:24:22.744781Z"},"links":{"citing_paper":"/paper/2507.02825"},"observation_digest":"sha256:dcf64f162b6be2bf76ac480a25d1f04bdfbecb6e237e8d49bb24a727ec78378c","observation_id":"add6f924-0680-4f83-94e7-1440ddcdaba8","resolution":{"observed_at":"2026-08-06T20:24:22.744781Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:24:22.816217Z","title":"Swe-bench: Can language models resolve real-world github issues? In ICLR, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.02825","last_updated":"2025-08-07T06:58:08Z","snapshot_observed_at":"2026-08-17T22:49:53.742436Z","submitted_at":"2025-07-03T17:35:31Z","title":"Establishing Best Practices for Building Rigorous Agentic Benchmarks","version":5},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-08-06T20:24:22.816217Z"},"links":{"citing_paper":"/paper/2507.02825"},"observation_digest":"sha256:b17595d2cfc05f45fb07f0464d9a41286e8309323c170669fb236b1f9a832250","observation_id":"6739a092-f9f4-4199-be3a-b4940e5bab8f","resolution":{"observed_at":"2026-08-06T20:24:22.816217Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:24:22.883514Z","title":"Swe-bench verified leaderboard, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.02825","last_updated":"2025-08-07T06:58:08Z","snapshot_observed_at":"2026-08-17T22:49:53.742436Z","submitted_at":"2025-07-03T17:35:31Z","title":"Establishing Best Practices for Building Rigorous Agentic Benchmarks","version":5},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-06T20:24:22.883514Z"},"links":{"citing_paper":"/paper/2507.02825"},"observation_digest":"sha256:70ef39f6b790c06af71e156bf8f364f5d8c8ea00b479d84fb88a6ce41aa30b05","observation_id":"79952336-47d2-4a0e-a240-a8843eeec48a","resolution":{"observed_at":"2026-08-06T20:24:22.883514Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2407.01502","last_updated":"2024-07-01T17:48:14Z","snapshot_observed_at":"2026-08-19T14:24:18.304488Z","submitted_at":"2024-07-01T17:48:14Z","title":"AI Agents That Matter","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2407.01502","snapshot_observed_at":"2026-08-06T20:24:22.949724Z","title":"Ai agents that matter","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.02825","last_updated":"2025-08-07T06:58:08Z","snapshot_observed_at":"2026-08-17T22:49:53.742436Z","submitted_at":"2025-07-03T17:35:31Z","title":"Establishing Best Practices for Building Rigorous Agentic Benchmarks","version":5},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-06T20:24:22.949724Z"},"links":{"cited_paper":"/paper/2407.01502","citing_paper":"/paper/2507.02825"},"observation_digest":"sha256:d5aacaea55b5f57dab654eb305969ae707f8be93bf092784fc7d47bf94bec703","observation_id":"9be6db01-506c-4190-b663-1296790ee689","resolution":{"observed_at":"2026-08-06T20:24:22.949724Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:24:23.020325Z","title":"Gemini 2.0 is now available to everyone, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.02825","last_updated":"2025-08-07T06:58:08Z","snapshot_observed_at":"2026-08-17T22:49:53.742436Z","submitted_at":"2025-07-03T17:35:31Z","title":"Establishing Best Practices for Building Rigorous Agentic Benchmarks","version":5},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-06T20:24:23.020325Z"},"links":{"citing_paper":"/paper/2507.02825"},"observation_digest":"sha256:a027104d1701ccec2e6c24cd89c0a4a64d6d1cc7e9dd60b98d4e046afbb64ca3","observation_id":"122680c8-e8f9-4bcd-9ecf-6ea0d54319ac","resolution":{"observed_at":"2026-08-06T20:24:23.020325Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:24:23.130120Z","title":"Math-verify, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.02825","last_updated":"2025-08-07T06:58:08Z","snapshot_observed_at":"2026-08-17T22:49:53.742436Z","submitted_at":"2025-07-03T17:35:31Z","title":"Establishing Best Practices for Building Rigorous Agentic Benchmarks","version":5},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-08-06T20:24:23.130120Z"},"links":{"citing_paper":"/paper/2507.02825"},"observation_digest":"sha256:9009915e6905d3ed5e7a2b382c5b7e6601c8771acd847ce229f87e289a1759d9","observation_id":"1c028daa-0318-4b18-9949-f93bc0f69941","resolution":{"observed_at":"2026-08-06T20:24:23.130120Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:24:23.259606Z","title":"The ai cuda engineer: Agentic cuda kernel discovery, optimization and composition","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.02825","last_updated":"2025-08-07T06:58:08Z","snapshot_observed_at":"2026-08-17T22:49:53.742436Z","submitted_at":"2025-07-03T17:35:31Z","title":"Establishing Best Practices for Building Rigorous Agentic Benchmarks","version":5},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-08-06T20:24:23.259606Z"},"links":{"citing_paper":"/paper/2507.02825"},"observation_digest":"sha256:1533b004dcf94b4eaeebdd42900c7902a0f9ff49cf91d8a363d2cc262506982f","observation_id":"a4b01508-7d1a-4ad4-af77-594e0d319e56","resolution":{"observed_at":"2026-08-06T20:24:23.259606Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:24:23.335796Z","title":"Challenges of end-to-end testing with selenium webdriver and how to face them: A survey","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.02825","last_updated":"2025-08-07T06:58:08Z","snapshot_observed_at":"2026-08-17T22:49:53.742436Z","submitted_at":"2025-07-03T17:35:31Z","title":"Establishing Best Practices for Building Rigorous Agentic Benchmarks","version":5},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-06T20:24:23.335796Z"},"links":{"citing_paper":"/paper/2507.02825"},"observation_digest":"sha256:543df87c05d2f56f8b00c70dc520477ea3ceb6ae7bbaaf112d6be6947b40580f","observation_id":"592213bb-34ee-436c-ab07-fbb055fc2201","resolution":{"observed_at":"2026-08-06T20:24:23.335796Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:24:23.446008Z","title":"Can llm already serve as a database interface? a big bench for large-scale database grounded text-to-sqls","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.02825","last_updated":"2025-08-07T06:58:08Z","snapshot_observed_at":"2026-08-17T22:49:53.742436Z","submitted_at":"2025-07-03T17:35:31Z","title":"Establishing Best Practices for Building Rigorous Agentic Benchmarks","version":5},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-08-06T20:24:23.446008Z"},"links":{"citing_paper":"/paper/2507.02825"},"observation_digest":"sha256:ad224b18abe1a95c8b3b3bd052dd01ef079bb74eaa75551739693419ecdb11e5","observation_id":"8abab65e-cefc-4dcb-8bcf-2afa33484a64","resolution":{"observed_at":"2026-08-06T20:24:23.446008Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:24:23.563205Z","title":"Leveraging large language models for nlg evaluation: Advances and challenges","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.02825","last_updated":"2025-08-07T06:58:08Z","snapshot_observed_at":"2026-08-17T22:49:53.742436Z","submitted_at":"2025-07-03T17:35:31Z","title":"Establishing Best Practices for Building Rigorous Agentic Benchmarks","version":5},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-08-06T20:24:23.563205Z"},"links":{"citing_paper":"/paper/2507.02825"},"observation_digest":"sha256:da374c7620a53dc9cd0329b83897e301e6c9ecea9cb48fc579fe1563c78db121","observation_id":"56d94836-4c1c-4215-a5d3-0db83908b2d9","resolution":{"observed_at":"2026-08-06T20:24:23.563205Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:24:23.625816Z","title":"Let’s verify step by step","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.02825","last_updated":"2025-08-07T06:58:08Z","snapshot_observed_at":"2026-08-17T22:49:53.742436Z","submitted_at":"2025-07-03T17:35:31Z","title":"Establishing Best Practices for Building Rigorous Agentic Benchmarks","version":5},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-08-06T20:24:23.625816Z"},"links":{"citing_paper":"/paper/2507.02825"},"observation_digest":"sha256:cd946489ab222bcb289efe3e4f0ddc48860a219e1f0edad6bf1bdd96b842e824","observation_id":"7754d152-ee3c-4cbd-9714-5689e29767f3","resolution":{"observed_at":"2026-08-06T20:24:23.625816Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:24:23.739707Z","title":"Swiftsage: A generative agent with fast and slow thinking for complex interactive tasks","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.02825","last_updated":"2025-08-07T06:58:08Z","snapshot_observed_at":"2026-08-17T22:49:53.742436Z","submitted_at":"2025-07-03T17:35:31Z","title":"Establishing Best Practices for Building Rigorous Agentic Benchmarks","version":5},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-08-06T20:24:23.739707Z"},"links":{"citing_paper":"/paper/2507.02825"},"observation_digest":"sha256:088cc382c86743cb18ac70fd1d627255fce0d2bcfedd1f587c0c158c4c8b2db4","observation_id":"9dce8c3f-afdf-480f-b6b9-62f1acf23957","resolution":{"observed_at":"2026-08-06T20:24:23.739707Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.09880","last_updated":"2024-10-14T02:11:29Z","snapshot_observed_at":"2026-08-20T13:45:09.934614Z","submitted_at":"2024-02-15T11:08:10Z","title":"Inadequacies of Large Language Model Benchmarks in the Era of Generative Artificial Intelligence","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.09880","snapshot_observed_at":"2026-08-06T20:24:23.835725Z","title":"Inadequacies of large language model benchmarks in the era of generative artificial intelligence","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.02825","last_updated":"2025-08-07T06:58:08Z","snapshot_observed_at":"2026-08-17T22:49:53.742436Z","submitted_at":"2025-07-03T17:35:31Z","title":"Establishing Best Practices for Building Rigorous Agentic Benchmarks","version":5},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-08-06T20:24:23.835725Z"},"links":{"cited_paper":"/paper/2402.09880","citing_paper":"/paper/2507.02825"},"observation_digest":"sha256:213ad8d15b07954ba99aa849499473fddfe7fc33e067edc773e9a798853ca86f","observation_id":"9c272ff5-c8e7-4443-82c4-911beb380184","resolution":{"observed_at":"2026-08-06T20:24:23.835725Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:24:23.966371Z","title":"Introducing llama 3.1: Our most capable models to date, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.02825","last_updated":"2025-08-07T06:58:08Z","snapshot_observed_at":"2026-08-17T22:49:53.742436Z","submitted_at":"2025-07-03T17:35:31Z","title":"Establishing Best Practices for Building Rigorous Agentic Benchmarks","version":5},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-08-06T20:24:23.966371Z"},"links":{"citing_paper":"/paper/2507.02825"},"observation_digest":"sha256:472a4c85c9911682d3c994c51f2b6b427aa9804d66f92042bcf9640d4d628936","observation_id":"e890901b-c91f-4e39-acfd-1e255458360b","resolution":{"observed_at":"2026-08-06T20:24:23.966371Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:24:24.051388Z","title":"Llama 3.2: Revolutionizing edge ai and vision with open, customizable models, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.02825","last_updated":"2025-08-07T06:58:08Z","snapshot_observed_at":"2026-08-17T22:49:53.742436Z","submitted_at":"2025-07-03T17:35:31Z","title":"Establishing Best Practices for Building Rigorous Agentic Benchmarks","version":5},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-06T20:24:24.051388Z"},"links":{"citing_paper":"/paper/2507.02825"},"observation_digest":"sha256:1d8e2c6e01b646660967b4e7e88402a27ac23f2239368592ef75228de006f5f7","observation_id":"9f9b52ed-450a-4b93-9159-ef8f1d1a6967","resolution":{"observed_at":"2026-08-06T20:24:24.051388Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:24:24.145765Z","title":"Evaluating language-model agents on realistic autonomous tasks, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.02825","last_updated":"2025-08-07T06:58:08Z","snapshot_observed_at":"2026-08-17T22:49:53.742436Z","submitted_at":"2025-07-03T17:35:31Z","title":"Establishing Best Practices for Building Rigorous Agentic Benchmarks","version":5},"reference_index":44,"source":"pdf_text","source_observed_at":"2026-08-06T20:24:24.145765Z"},"links":{"citing_paper":"/paper/2507.02825"},"observation_digest":"sha256:b666c9abc2a9d40ebceda6e35c47a90cb63f4b83baa7ebb7626ee7973b9aab8f","observation_id":"ec503b05-2bc5-4edb-9c9c-ed9b724b7cb4","resolution":{"observed_at":"2026-08-06T20:24:24.145765Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:24:24.235381Z","title":"Example protocol for running an ai agent evaluation, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.02825","last_updated":"2025-08-07T06:58:08Z","snapshot_observed_at":"2026-08-17T22:49:53.742436Z","submitted_at":"2025-07-03T17:35:31Z","title":"Establishing Best Practices for Building Rigorous Agentic Benchmarks","version":5},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-06T20:24:24.235381Z"},"links":{"citing_paper":"/paper/2507.02825"},"observation_digest":"sha256:3fc74783951d7c3ace2a26c7251b0ffb9d39a0535e75765da2ccbc6c7db0d146","observation_id":"28c66277-51aa-4e67-8bf9-db61072802e6","resolution":{"observed_at":"2026-08-06T20:24:24.235381Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:24:24.324881Z","title":"Measuring automated kernel engineering, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.02825","last_updated":"2025-08-07T06:58:08Z","snapshot_observed_at":"2026-08-17T22:49:53.742436Z","submitted_at":"2025-07-03T17:35:31Z","title":"Establishing Best Practices for Building Rigorous Agentic Benchmarks","version":5},"reference_index":46,"source":"pdf_text","source_observed_at":"2026-08-06T20:24:24.324881Z"},"links":{"citing_paper":"/paper/2507.02825"},"observation_digest":"sha256:6141a59700d6d30e26f31c8b1e5ce3b4646acd729334430c899297409d0f3cc6","observation_id":"93b5332d-9b40-4d92-ab18-7ae10a98af38","resolution":{"observed_at":"2026-08-06T20:24:24.324881Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:24:24.418642Z","title":"Gaia: a benchmark for general ai assistants","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.02825","last_updated":"2025-08-07T06:58:08Z","snapshot_observed_at":"2026-08-17T22:49:53.742436Z","submitted_at":"2025-07-03T17:35:31Z","title":"Establishing Best Practices for Building Rigorous Agentic Benchmarks","version":5},"reference_index":47,"source":"pdf_text","source_observed_at":"2026-08-06T20:24:24.418642Z"},"links":{"citing_paper":"/paper/2507.02825"},"observation_digest":"sha256:b6a6854371656649a4a26a386c97afae49661cf6fea596a441d01d0e1f6c125b","observation_id":"a356301a-7ba5-4f73-98eb-a48ec47c1579","resolution":{"observed_at":"2026-08-06T20:24:24.418642Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.12115","last_updated":"2025-05-29T23:07:34Z","snapshot_observed_at":"2026-08-18T15:49:49.317002Z","submitted_at":"2025-02-17T18:41:16Z","title":"SWE-Lancer: Can Frontier LLMs Earn $1 Million from Real-World Freelance Software Engineering?","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.12115","snapshot_observed_at":"2026-08-06T20:24:24.518781Z","title":"Swe-lancer: Can frontier llms earn $1 million from real-world freelance software engineering? arXiv preprint arXiv:2502.12115, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.02825","last_updated":"2025-08-07T06:58:08Z","snapshot_observed_at":"2026-08-17T22:49:53.742436Z","submitted_at":"2025-07-03T17:35:31Z","title":"Establishing Best Practices for Building Rigorous Agentic Benchmarks","version":5},"reference_index":48,"source":"pdf_text","source_observed_at":"2026-08-06T20:24:24.518781Z"},"links":{"cited_paper":"/paper/2502.12115","citing_paper":"/paper/2507.02825"},"observation_digest":"sha256:7f4657b15ec91f9c03340e58b8af9114c2e0d5b59e7a644ce89232f2f217f34c","observation_id":"6d5570e0-69c3-463d-bcd0-5dcb2af1dcfe","resolution":{"observed_at":"2026-08-06T20:24:24.518781Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:24:24.571947Z","title":"Mixtral large 2, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.02825","last_updated":"2025-08-07T06:58:08Z","snapshot_observed_at":"2026-08-17T22:49:53.742436Z","submitted_at":"2025-07-03T17:35:31Z","title":"Establishing Best Practices for Building Rigorous Agentic Benchmarks","version":5},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-06T20:24:24.571947Z"},"links":{"citing_paper":"/paper/2507.02825"},"observation_digest":"sha256:cae296fe2cee5cc105b2847f15c998df299a19470d4ca78c654b775ddf3e768e","observation_id":"78279a50-f525-4cc2-a122-712c5c892bb7","resolution":{"observed_at":"2026-08-06T20:24:24.571947Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:24:24.632617Z","title":"Preparedness framework (beta), 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.02825","last_updated":"2025-08-07T06:58:08Z","snapshot_observed_at":"2026-08-17T22:49:53.742436Z","submitted_at":"2025-07-03T17:35:31Z","title":"Establishing Best Practices for Building Rigorous Agentic Benchmarks","version":5},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-06T20:24:24.632617Z"},"links":{"citing_paper":"/paper/2507.02825"},"observation_digest":"sha256:fc88c7f35899a1afbb50055051553b1c81901c43c2fa022919b991ba5834be76","observation_id":"d6f9dd39-5c42-497a-98f3-d3b1cb4c5ea5","resolution":{"observed_at":"2026-08-06T20:24:24.632617Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:24:40.018286Z","title":"Gpt-4o system card, 2024","venue":null,"work_id":"d77f1037-96f3-4b67-b5b8-ca79316f7c26","year":2024},"citing_paper":{"arxiv_id":"2507.02825","last_updated":"2025-08-07T06:58:08Z","snapshot_observed_at":"2026-08-17T22:49:53.742436Z","submitted_at":"2025-07-03T17:35:31Z","title":"Establishing Best Practices for Building Rigorous Agentic Benchmarks","version":5},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-06T20:24:24.756415Z"},"links":{"citing_paper":"/paper/2507.02825"},"observation_digest":"sha256:2344cdebd9a9353ca96dd60ffc2af1964d283f7cb12e0af862bec5adad2306e3","observation_id":"1f21e953-864c-49ee-ab92-b677d6ad2eeb","resolution":{"observed_at":"2026-08-06T20:24:40.110959Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:24:24.825783Z","title":"Openai o1 system card, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.02825","last_updated":"2025-08-07T06:58:08Z","snapshot_observed_at":"2026-08-17T22:49:53.742436Z","submitted_at":"2025-07-03T17:35:31Z","title":"Establishing Best Practices for Building Rigorous Agentic Benchmarks","version":5},"reference_index":52,"source":"pdf_text","source_observed_at":"2026-08-06T20:24:24.825783Z"},"links":{"citing_paper":"/paper/2507.02825"},"observation_digest":"sha256:9806c1002243f5e86b79cfa83c30c17d8aa47603e5c09c819fbcd12c01f31455","observation_id":"1009fd69-fc03-4e62-ba60-ceb30aa5ad3a","resolution":{"observed_at":"2026-08-06T20:24:24.825783Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:24:39.858951Z","title":"Openai o1-mini, 2024","venue":null,"work_id":"5bf0e3c7-f5d9-4f19-ae0d-cb93a7c575c0","year":2024},"citing_paper":{"arxiv_id":"2507.02825","last_updated":"2025-08-07T06:58:08Z","snapshot_observed_at":"2026-08-17T22:49:53.742436Z","submitted_at":"2025-07-03T17:35:31Z","title":"Establishing Best Practices for Building Rigorous Agentic Benchmarks","version":5},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-08-06T20:24:24.937917Z"},"links":{"citing_paper":"/paper/2507.02825"},"observation_digest":"sha256:2fde3d64120f600b85881cb7da9ab2561a6815307599ce2deebcb3e38efe51c5","observation_id":"54095ed0-51e5-498c-8269-f597bd1b563a","resolution":{"observed_at":"2026-08-06T20:24:39.935710Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:24:39.651196Z","title":"Gpt-4o mini: advancing cost-efficient intelligence, 2025","venue":null,"work_id":"6dbd5f6a-a2dc-4cea-a562-ec9fa685b57e","year":2025},"citing_paper":{"arxiv_id":"2507.02825","last_updated":"2025-08-07T06:58:08Z","snapshot_observed_at":"2026-08-17T22:49:53.742436Z","submitted_at":"2025-07-03T17:35:31Z","title":"Establishing Best Practices for Building Rigorous Agentic Benchmarks","version":5},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-08-06T20:24:25.016609Z"},"links":{"citing_paper":"/paper/2507.02825"},"observation_digest":"sha256:f691f597f4ab044b463348b41f5c127e0aefb66ce95d040eba4214dbf5351cc3","observation_id":"d2e2abee-bbef-4aa6-9b39-bb8f5c230a7b","resolution":{"observed_at":"2026-08-06T20:24:39.740345Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:24:39.479968Z","title":"Computer-user agent, 2025","venue":null,"work_id":"ca60ba0f-f974-4fa6-a830-b307b932c5f4","year":2025},"citing_paper":{"arxiv_id":"2507.02825","last_updated":"2025-08-07T06:58:08Z","snapshot_observed_at":"2026-08-17T22:49:53.742436Z","submitted_at":"2025-07-03T17:35:31Z","title":"Establishing Best Practices for Building Rigorous Agentic Benchmarks","version":5},"reference_index":55,"source":"pdf_text","source_observed_at":"2026-08-06T20:24:25.134135Z"},"links":{"citing_paper":"/paper/2507.02825"},"observation_digest":"sha256:e302045633c7e8c1b1ce0a37ef12a4b7850e162c944d60fb468862d296b0ebf9","observation_id":"aa6d0ab7-267f-4046-9da4-9fa7af0b1d45","resolution":{"observed_at":"2026-08-06T20:24:39.565147Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:24:25.231381Z","title":"Introducing deep research, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.02825","last_updated":"2025-08-07T06:58:08Z","snapshot_observed_at":"2026-08-17T22:49:53.742436Z","submitted_at":"2025-07-03T17:35:31Z","title":"Establishing Best Practices for Building Rigorous Agentic Benchmarks","version":5},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-08-06T20:24:25.231381Z"},"links":{"citing_paper":"/paper/2507.02825"},"observation_digest":"sha256:f4c93420ce3f9950a17595fc34e95a259318a14bec4c498a928576071ad41e24","observation_id":"2c55a375-688b-4b5b-9074-2484f531c990","resolution":{"observed_at":"2026-08-06T20:24:25.231381Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:24:39.305810Z","title":"Introducing gpt-4.5, 2025","venue":null,"work_id":"cc9201c2-4f30-4c57-8f8c-f7ed8b029c8a","year":2025},"citing_paper":{"arxiv_id":"2507.02825","last_updated":"2025-08-07T06:58:08Z","snapshot_observed_at":"2026-08-17T22:49:53.742436Z","submitted_at":"2025-07-03T17:35:31Z","title":"Establishing Best Practices for Building Rigorous Agentic Benchmarks","version":5},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-08-06T20:24:25.310506Z"},"links":{"citing_paper":"/paper/2507.02825"},"observation_digest":"sha256:69832a4a660d80f4b8a357360a7c037d9aa3ccd5f30feeef37e5b21af26c9827","observation_id":"dbf002cd-0fce-4028-a4b0-39c10a8248ef","resolution":{"observed_at":"2026-08-06T20:24:39.386566Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:24:25.423058Z","title":"Openai o3-mini, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.02825","last_updated":"2025-08-07T06:58:08Z","snapshot_observed_at":"2026-08-17T22:49:53.742436Z","submitted_at":"2025-07-03T17:35:31Z","title":"Establishing Best Practices for Building Rigorous Agentic Benchmarks","version":5},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-08-06T20:24:25.423058Z"},"links":{"citing_paper":"/paper/2507.02825"},"observation_digest":"sha256:b23129b2394b82be52d41d893abffd409f1b65648a13aafee033fc1ad330d8ef","observation_id":"70b51f70-1084-44d1-a578-b7718c45ae58","resolution":{"observed_at":"2026-08-06T20:24:25.423058Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:24:39.124349Z","title":"Kernelbench: Can llms write efficient gpu kernels? In ICLR 2025 Third Workshop on Deep Learning for Code (Best paper award), 2025","venue":null,"work_id":"a60fefcf-7a4b-455b-b14f-246ee5f57175","year":2025},"citing_paper":{"arxiv_id":"2507.02825","last_updated":"2025-08-07T06:58:08Z","snapshot_observed_at":"2026-08-17T22:49:53.742436Z","submitted_at":"2025-07-03T17:35:31Z","title":"Establishing Best Practices for Building Rigorous Agentic Benchmarks","version":5},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-08-06T20:24:25.505512Z"},"links":{"citing_paper":"/paper/2507.02825"},"observation_digest":"sha256:2366380a60d9ed7f3313c3b85646aaa2977a3e0958c9bdab3f40ca37692554b8","observation_id":"e3e8824d-5612-467b-953a-585dfe728c2a","resolution":{"observed_at":"2026-08-06T20:24:39.204719Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:24:25.597919Z","title":"Bleu: a method for automatic evaluation of machine translation","venue":null,"work_id":null,"year":2002},"citing_paper":{"arxiv_id":"2507.02825","last_updated":"2025-08-07T06:58:08Z","snapshot_observed_at":"2026-08-17T22:49:53.742436Z","submitted_at":"2025-07-03T17:35:31Z","title":"Establishing Best Practices for Building Rigorous Agentic Benchmarks","version":5},"reference_index":60,"source":"pdf_text","source_observed_at":"2026-08-06T20:24:25.597919Z"},"links":{"citing_paper":"/paper/2507.02825"},"observation_digest":"sha256:954d437b737ef542cc27ac36530b523d8757cc686364a99e97ab8f498ea4d968","observation_id":"e6cdc13f-34aa-4119-84d2-630125c893d2","resolution":{"observed_at":"2026-08-06T20:24:25.597919Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:24:38.956374Z","title":"A survey of flaky tests","venue":null,"work_id":"ceceb536-38a4-4d1e-ad6d-d00327563df8","year":2021},"citing_paper":{"arxiv_id":"2507.02825","last_updated":"2025-08-07T06:58:08Z","snapshot_observed_at":"2026-08-17T22:49:53.742436Z","submitted_at":"2025-07-03T17:35:31Z","title":"Establishing Best Practices for Building Rigorous Agentic Benchmarks","version":5},"reference_index":61,"source":"pdf_text","source_observed_at":"2026-08-06T20:24:25.709063Z"},"links":{"citing_paper":"/paper/2507.02825"},"observation_digest":"sha256:c5cdacf1e566251f74856a981a87f2e67ad8bd797a248ed2cfb73700de276799","observation_id":"936e1ddd-4052-4ee0-884d-8a68c49e34e3","resolution":{"observed_at":"2026-08-06T20:24:39.047902Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.18538","last_updated":"2023-10-27T23:36:14Z","snapshot_observed_at":"2026-08-16T14:48:06.934536Z","submitted_at":"2023-10-27T23:36:14Z","title":"Evaluating Cross-Domain Text-to-SQL Models and Benchmarks","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.18538","snapshot_observed_at":"2026-08-06T20:24:25.823326Z","title":"Evaluating cross-domain text-to-sql models and benchmarks","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.02825","last_updated":"2025-08-07T06:58:08Z","snapshot_observed_at":"2026-08-17T22:49:53.742436Z","submitted_at":"2025-07-03T17:35:31Z","title":"Establishing Best Practices for Building Rigorous Agentic Benchmarks","version":5},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-08-06T20:24:25.823326Z"},"links":{"cited_paper":"/paper/2310.18538","citing_paper":"/paper/2507.02825"},"observation_digest":"sha256:e8386f61aca990ffd9a666a333192de5984ac3741b0d386fb468404ff86cd3cf","observation_id":"fba94f27-761d-4acd-a0d0-017ad16ac8c7","resolution":{"observed_at":"2026-08-06T20:24:25.823326Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2501.01257","last_updated":"2025-01-03T16:36:12Z","snapshot_observed_at":"2026-08-18T05:16:26.684195Z","submitted_at":"2025-01-02T13:49:00Z","title":"CodeElo: Benchmarking Competition-level Code Generation of LLMs with Human-comparable Elo Ratings","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2501.01257","snapshot_observed_at":"2026-08-06T20:24:25.910315Z","title":"Codeelo: Benchmarking competition-level code generation of llms with human-comparable elo ratings","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.02825","last_updated":"2025-08-07T06:58:08Z","snapshot_observed_at":"2026-08-17T22:49:53.742436Z","submitted_at":"2025-07-03T17:35:31Z","title":"Establishing Best Practices for Building Rigorous Agentic Benchmarks","version":5},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-08-06T20:24:25.910315Z"},"links":{"cited_paper":"/paper/2501.01257","citing_paper":"/paper/2507.02825"},"observation_digest":"sha256:02ce786ad2e12295c1e3a4989b8e27b0d47b01883e1442a2674357e952470947","observation_id":"99e58958-ce9a-4adf-b53e-55b5051e5786","resolution":{"observed_at":"2026-08-06T20:24:25.910315Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:24:38.744270Z","title":"Ai and the everything in the whole wide world benchmark","venue":null,"work_id":"07dc7bdc-1855-404d-bdbe-205157331ca9","year":null},"citing_paper":{"arxiv_id":"2507.02825","last_updated":"2025-08-07T06:58:08Z","snapshot_observed_at":"2026-08-17T22:49:53.742436Z","submitted_at":"2025-07-03T17:35:31Z","title":"Establishing Best Practices for Building Rigorous Agentic Benchmarks","version":5},"reference_index":64,"source":"pdf_text","source_observed_at":"2026-08-06T20:24:26.004695Z"},"links":{"citing_paper":"/paper/2507.02825"},"observation_digest":"sha256:a959d6de9b0dd4e3f7f9bb86deb7025840e771e180f6f4993853afe5128237bc","observation_id":"0f260b87-08dc-41ae-bf50-586b86aefa6f","resolution":{"observed_at":"2026-08-06T20:24:38.849661Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:24:38.559914Z","title":"Betterbench: Assessing ai benchmarks, uncovering issues, and establishing best practices","venue":null,"work_id":"7ebbe0d7-13f3-4579-80d9-bae9eb3417c9","year":2024},"citing_paper":{"arxiv_id":"2507.02825","last_updated":"2025-08-07T06:58:08Z","snapshot_observed_at":"2026-08-17T22:49:53.742436Z","submitted_at":"2025-07-03T17:35:31Z","title":"Establishing Best Practices for Building Rigorous Agentic Benchmarks","version":5},"reference_index":65,"source":"pdf_text","source_observed_at":"2026-08-06T20:24:26.112012Z"},"links":{"citing_paper":"/paper/2507.02825"},"observation_digest":"sha256:abfffd7bf70917874d5821039f8d38252c2d5a364653a91c17a40bf9056b47f0","observation_id":"7a9bd51b-a180-4b3e-a5cb-1f3ab7cd18ad","resolution":{"observed_at":"2026-08-06T20:24:38.647334Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:24:38.359303Z","title":"Analysis and testing of web applications","venue":null,"work_id":"910357d3-b764-4587-95be-01b4a258c2f0","year":2001},"citing_paper":{"arxiv_id":"2507.02825","last_updated":"2025-08-07T06:58:08Z","snapshot_observed_at":"2026-08-17T22:49:53.742436Z","submitted_at":"2025-07-03T17:35:31Z","title":"Establishing Best Practices for Building Rigorous Agentic Benchmarks","version":5},"reference_index":66,"source":"pdf_text","source_observed_at":"2026-08-06T20:24:26.208330Z"},"links":{"citing_paper":"/paper/2507.02825"},"observation_digest":"sha256:ce66270eeb1a2a0af9a2028d31818570fbc8c8f0e7a09706a4ebe443ae6cbd81","observation_id":"1c2b8a5e-347e-4952-a339-46d484dbfae8","resolution":{"observed_at":"2026-08-06T20:24:38.455331Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:24:38.123633Z","title":"A survey of unit testing practices","venue":null,"work_id":"b253581e-f521-4072-babd-dbb6ee65e096","year":2006},"citing_paper":{"arxiv_id":"2507.02825","last_updated":"2025-08-07T06:58:08Z","snapshot_observed_at":"2026-08-17T22:49:53.742436Z","submitted_at":"2025-07-03T17:35:31Z","title":"Establishing Best Practices for Building Rigorous Agentic Benchmarks","version":5},"reference_index":67,"source":"pdf_text","source_observed_at":"2026-08-06T20:24:26.295615Z"},"links":{"citing_paper":"/paper/2507.02825"},"observation_digest":"sha256:2127c5a568960b1bf8c2ce4ad207a87498f1d99af79f2c5de021ed9874c8b209","observation_id":"8bfbeba8-34bc-4404-8c84-a03a88d22991","resolution":{"observed_at":"2026-08-06T20:24:38.245734Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:24:37.938212Z","title":"Reflexion: Language agents with verbal reinforcement learning.Advances in Neural Information Processing Systems, 36:8634–8652, 2023","venue":null,"work_id":"87b67d90-56b0-474b-8487-64ad05d0e612","year":2023},"citing_paper":{"arxiv_id":"2507.02825","last_updated":"2025-08-07T06:58:08Z","snapshot_observed_at":"2026-08-17T22:49:53.742436Z","submitted_at":"2025-07-03T17:35:31Z","title":"Establishing Best Practices for Building Rigorous Agentic Benchmarks","version":5},"reference_index":68,"source":"pdf_text","source_observed_at":"2026-08-06T20:24:26.364230Z"},"links":{"citing_paper":"/paper/2507.02825"},"observation_digest":"sha256:5fc95b175e230a265eafee788936088f0caf5e8aa1ab27fae2aa620417f1aad9","observation_id":"adff7b59-e754-4dd2-b0e6-4130982c848a","resolution":{"observed_at":"2026-08-06T20:24:38.017958Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:24:37.717797Z","title":"Strengthening ai agent hijacking eval- uations, 2025","venue":null,"work_id":"9dcac5b6-78d7-480c-8dbf-5a09321e20b5","year":2025},"citing_paper":{"arxiv_id":"2507.02825","last_updated":"2025-08-07T06:58:08Z","snapshot_observed_at":"2026-08-17T22:49:53.742436Z","submitted_at":"2025-07-03T17:35:31Z","title":"Establishing Best Practices for Building Rigorous Agentic Benchmarks","version":5},"reference_index":69,"source":"pdf_text","source_observed_at":"2026-08-06T20:24:26.483123Z"},"links":{"citing_paper":"/paper/2507.02825"},"observation_digest":"sha256:94a12c9bd333f86df8b5a5266d630db575e8fcee428e5a60dc14e901b2868fa9","observation_id":"100925d5-acb5-4e73-a608-94591c4751d9","resolution":{"observed_at":"2026-08-06T20:24:37.816267Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:24:26.559726Z","title":"Inference scaling flaws: The limits of llm resampling with imperfect verifiers","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.02825","last_updated":"2025-08-07T06:58:08Z","snapshot_observed_at":"2026-08-17T22:49:53.742436Z","submitted_at":"2025-07-03T17:35:31Z","title":"Establishing Best Practices for Building Rigorous Agentic Benchmarks","version":5},"reference_index":70,"source":"pdf_text","source_observed_at":"2026-08-06T20:24:26.559726Z"},"links":{"citing_paper":"/paper/2507.02825"},"observation_digest":"sha256:fe95008e1f35cc2352c65e3a375c61059013e54ee1af58dd14714b3dbafd84e0","observation_id":"e2b3a539-22ca-43d0-911a-53dc5759bb66","resolution":{"observed_at":"2026-08-06T20:24:26.559726Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:24:37.500512Z","title":"Worldcoder, a model-based llm agent: Building world models by writing code and interacting with the environment","venue":null,"work_id":"b654e17e-9abe-43c7-b824-30985a8de948","year":2024},"citing_paper":{"arxiv_id":"2507.02825","last_updated":"2025-08-07T06:58:08Z","snapshot_observed_at":"2026-08-17T22:49:53.742436Z","submitted_at":"2025-07-03T17:35:31Z","title":"Establishing Best Practices for Building Rigorous Agentic Benchmarks","version":5},"reference_index":71,"source":"pdf_text","source_observed_at":"2026-08-06T20:24:26.685912Z"},"links":{"citing_paper":"/paper/2507.02825"},"observation_digest":"sha256:d11e23feee622e1db9293df766385c40e34ea0434f529bbdff3e1ef6929fd7bb","observation_id":"42f6f29f-a16c-448e-bbd4-d2fa8b248c44","resolution":{"observed_at":"2026-08-06T20:24:37.604621Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:24:37.182065Z","title":"End-to-end integration testing design","venue":null,"work_id":"2a5f8ff6-3930-40e2-b266-951072acc434","year":2001},"citing_paper":{"arxiv_id":"2507.02825","last_updated":"2025-08-07T06:58:08Z","snapshot_observed_at":"2026-08-17T22:49:53.742436Z","submitted_at":"2025-07-03T17:35:31Z","title":"Establishing Best Practices for Building Rigorous Agentic Benchmarks","version":5},"reference_index":72,"source":"pdf_text","source_observed_at":"2026-08-06T20:24:26.752737Z"},"links":{"citing_paper":"/paper/2507.02825"},"observation_digest":"sha256:f9b639d552cdd31197c163b66a0a09a5cc5ee615ad7a6e6d0825730fab338a47","observation_id":"7b3638e3-ecb7-4851-8325-2d8cc4dc018b","resolution":{"observed_at":"2026-08-06T20:24:37.334722Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:24:36.857057Z","title":"From imagenet to image classification: Contextualizing progress on benchmarks","venue":null,"work_id":"d8b21e2b-df3e-43b4-9111-dbd456e33f44","year":2020},"citing_paper":{"arxiv_id":"2507.02825","last_updated":"2025-08-07T06:58:08Z","snapshot_observed_at":"2026-08-17T22:49:53.742436Z","submitted_at":"2025-07-03T17:35:31Z","title":"Establishing Best Practices for Building Rigorous Agentic Benchmarks","version":5},"reference_index":73,"source":"pdf_text","source_observed_at":"2026-08-06T20:24:26.832387Z"},"links":{"citing_paper":"/paper/2507.02825"},"observation_digest":"sha256:f80861e71a9ef6b84fd081f8eaf05fcda9ea7e5a57790e4f2896937aa0a16cfe","observation_id":"a43f2b5b-cef1-4b2c-8f52-c4192b29f12c","resolution":{"observed_at":"2026-08-06T20:24:37.019808Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.12241","last_updated":"2024-05-13T20:46:10Z","snapshot_observed_at":"2026-08-20T19:31:52.934922Z","submitted_at":"2024-04-18T15:01:00Z","title":"Introducing v0.5 of the AI Safety Benchmark from MLCommons","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2404.12241","snapshot_observed_at":"2026-08-06T20:24:26.965009Z","title":"Introducing v0","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.02825","last_updated":"2025-08-07T06:58:08Z","snapshot_observed_at":"2026-08-17T22:49:53.742436Z","submitted_at":"2025-07-03T17:35:31Z","title":"Establishing Best Practices for Building Rigorous Agentic Benchmarks","version":5},"reference_index":74,"source":"pdf_text","source_observed_at":"2026-08-06T20:24:26.965009Z"},"links":{"cited_paper":"/paper/2404.12241","citing_paper":"/paper/2507.02825"},"observation_digest":"sha256:476e26d17229f20be3c6984fa0968a86830b400915877b0aace08b2caa55e0b0","observation_id":"fb92c734-fafe-4a5f-ba81-1a2c4dfd05a2","resolution":{"observed_at":"2026-08-06T20:24:26.965009Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:24:36.612907Z","title":"Evaluate & evaluation on the hub: Better best practices for data and model measurements","venue":null,"work_id":"49d07336-2b8a-47e6-936e-8983debf8600","year":2022},"citing_paper":{"arxiv_id":"2507.02825","last_updated":"2025-08-07T06:58:08Z","snapshot_observed_at":"2026-08-17T22:49:53.742436Z","submitted_at":"2025-07-03T17:35:31Z","title":"Establishing Best Practices for Building Rigorous Agentic Benchmarks","version":5},"reference_index":75,"source":"pdf_text","source_observed_at":"2026-08-06T20:24:27.071959Z"},"links":{"citing_paper":"/paper/2507.02825"},"observation_digest":"sha256:ef021fe19fb7e88a975a3bf20de7942c16aa454cc8932548e94a660347527297","observation_id":"0a0fea27-b91a-4cf4-abb1-cb3662ed0c6d","resolution":{"observed_at":"2026-08-06T20:24:36.730961Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:24:36.409636Z","title":"Structured testing: A testing methodology using the cyclomatic complexity metric, volume 500","venue":null,"work_id":"3713893a-224a-4459-861d-09dffc82a66b","year":1996},"citing_paper":{"arxiv_id":"2507.02825","last_updated":"2025-08-07T06:58:08Z","snapshot_observed_at":"2026-08-17T22:49:53.742436Z","submitted_at":"2025-07-03T17:35:31Z","title":"Establishing Best Practices for Building Rigorous Agentic Benchmarks","version":5},"reference_index":76,"source":"pdf_text","source_observed_at":"2026-08-06T20:24:27.172490Z"},"links":{"citing_paper":"/paper/2507.02825"},"observation_digest":"sha256:76d71e472257322812919e8aff958dd98d509a50d0c810181ca7d6b6377dfa1f","observation_id":"bfecfd29-8d68-491c-9243-8e1afe8547f4","resolution":{"observed_at":"2026-08-06T20:24:36.494484Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.04368","last_updated":"2024-11-07T01:58:42Z","snapshot_observed_at":"2026-08-17T10:08:57.374438Z","submitted_at":"2024-11-07T01:58:42Z","title":"Measuring short-form factuality in large language models","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.04368","snapshot_observed_at":"2026-08-06T20:24:27.263615Z","title":"Measuring short-form factuality in large language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.02825","last_updated":"2025-08-07T06:58:08Z","snapshot_observed_at":"2026-08-17T22:49:53.742436Z","submitted_at":"2025-07-03T17:35:31Z","title":"Establishing Best Practices for Building Rigorous Agentic Benchmarks","version":5},"reference_index":77,"source":"pdf_text","source_observed_at":"2026-08-06T20:24:27.263615Z"},"links":{"cited_paper":"/paper/2411.04368","citing_paper":"/paper/2507.02825"},"observation_digest":"sha256:8bf9a6642d94d2038b840f99576d70832748c54032aef1b4d65cd019bf663091","observation_id":"7e007b0c-0b86-4d42-94b7-84dd4f755e1e","resolution":{"observed_at":"2026-08-06T20:24:27.263615Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2406.19314","last_updated":"2025-04-18T19:36:00Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-06-27T16:47:42Z","title":"LiveBench: A Challenging, Contamination-Limited LLM Benchmark","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2406.19314","snapshot_observed_at":"2026-08-06T20:24:27.342525Z","title":"Livebench: A challenging, contamination-free llm benchmark","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.02825","last_updated":"2025-08-07T06:58:08Z","snapshot_observed_at":"2026-08-17T22:49:53.742436Z","submitted_at":"2025-07-03T17:35:31Z","title":"Establishing Best Practices for Building Rigorous Agentic Benchmarks","version":5},"reference_index":78,"source":"pdf_text","source_observed_at":"2026-08-06T20:24:27.342525Z"},"links":{"cited_paper":"/paper/2406.19314","citing_paper":"/paper/2507.02825"},"observation_digest":"sha256:75e0f6db8377edb8c2734152db8049229cd33d605a7f1e62171f77c7444aea50","observation_id":"1211ce8f-51ba-4ab2-9e95-f279f69a42df","resolution":{"observed_at":"2026-08-06T20:24:27.342525Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2411.15114","last_updated":"2025-05-27T03:32:23Z","snapshot_observed_at":"2026-08-12T16:49:37.896696Z","submitted_at":"2024-11-22T18:30:46Z","title":"RE-Bench: Evaluating frontier AI R&D capabilities of language model agents against human experts","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2411.15114","snapshot_observed_at":"2026-08-06T20:24:27.421486Z","title":"Re-bench: Evaluating frontier ai r&d capabilities of language model agents against human experts","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.02825","last_updated":"2025-08-07T06:58:08Z","snapshot_observed_at":"2026-08-17T22:49:53.742436Z","submitted_at":"2025-07-03T17:35:31Z","title":"Establishing Best Practices for Building Rigorous Agentic Benchmarks","version":5},"reference_index":79,"source":"pdf_text","source_observed_at":"2026-08-06T20:24:27.421486Z"},"links":{"cited_paper":"/paper/2411.15114","citing_paper":"/paper/2507.02825"},"observation_digest":"sha256:0836b2167a45975227309c928cf5461eaa545099c45ae5edf717c3afb87e56bc","observation_id":"36100b82-0a6a-453d-8fd8-7252d4bb2915","resolution":{"observed_at":"2026-08-06T20:24:27.421486Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2402.12243","last_updated":"2024-03-25T19:48:16Z","snapshot_observed_at":"2026-08-16T14:17:17.253731Z","submitted_at":"2024-02-19T15:58:15Z","title":"Understanding the Effects of Noise in Text-to-SQL: An Examination of the BIRD-Bench Benchmark","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2402.12243","snapshot_observed_at":"2026-08-06T20:24:27.539316Z","title":"Understanding the effects of noise in text-to-sql: an examination of the bird-bench benchmark","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.02825","last_updated":"2025-08-07T06:58:08Z","snapshot_observed_at":"2026-08-17T22:49:53.742436Z","submitted_at":"2025-07-03T17:35:31Z","title":"Establishing Best Practices for Building Rigorous Agentic Benchmarks","version":5},"reference_index":80,"source":"pdf_text","source_observed_at":"2026-08-06T20:24:27.539316Z"},"links":{"cited_paper":"/paper/2402.12243","citing_paper":"/paper/2507.02825"},"observation_digest":"sha256:6678fd62587cff48aeaecab4be0d81338c126544b3397099087cbc5679ad798a","observation_id":"f034431a-346d-4b7d-a505-30bc5b871529","resolution":{"observed_at":"2026-08-06T20:24:27.539316Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:24:36.239595Z","title":"Grok 2 beta release, 2024","venue":null,"work_id":"574360e6-22fb-42d7-8095-a13cc69a0c13","year":2024},"citing_paper":{"arxiv_id":"2507.02825","last_updated":"2025-08-07T06:58:08Z","snapshot_observed_at":"2026-08-17T22:49:53.742436Z","submitted_at":"2025-07-03T17:35:31Z","title":"Establishing Best Practices for Building Rigorous Agentic Benchmarks","version":5},"reference_index":81,"source":"pdf_text","source_observed_at":"2026-08-06T20:24:27.664743Z"},"links":{"citing_paper":"/paper/2507.02825"},"observation_digest":"sha256:a6d0db1bebbeba686964240b95db2b09a036140f3b87a22d64b03d7ad1c114ee","observation_id":"f10c0ec6-4971-4d89-a801-e30fc1a929be","resolution":{"observed_at":"2026-08-06T20:24:36.319408Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:24:36.085787Z","title":"Grok 3 beta — the age of reasoning agents, 2024","venue":null,"work_id":"972ea02d-2b91-4c30-90e8-c26efc7cc402","year":2024},"citing_paper":{"arxiv_id":"2507.02825","last_updated":"2025-08-07T06:58:08Z","snapshot_observed_at":"2026-08-17T22:49:53.742436Z","submitted_at":"2025-07-03T17:35:31Z","title":"Establishing Best Practices for Building Rigorous Agentic Benchmarks","version":5},"reference_index":82,"source":"pdf_text","source_observed_at":"2026-08-06T20:24:27.758444Z"},"links":{"citing_paper":"/paper/2507.02825"},"observation_digest":"sha256:9466ec1b2d2e9e1d884a6feecea86bb673c239012d69f28c1210d3b09e80a3f7","observation_id":"d6ebfa40-b878-4f3b-9a60-21e7ff5c8832","resolution":{"observed_at":"2026-08-06T20:24:36.151243Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:24:27.822200Z","title":"Osworld: Benchmarking multimodal agents for open-ended tasks in real computer environments","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.02825","last_updated":"2025-08-07T06:58:08Z","snapshot_observed_at":"2026-08-17T22:49:53.742436Z","submitted_at":"2025-07-03T17:35:31Z","title":"Establishing Best Practices for Building Rigorous Agentic Benchmarks","version":5},"reference_index":83,"source":"pdf_text","source_observed_at":"2026-08-06T20:24:27.822200Z"},"links":{"citing_paper":"/paper/2507.02825"},"observation_digest":"sha256:9e6b62b83b82e2877042319530144e682176cfa79deaa9f0e092b320eb870b2e","observation_id":"9792836c-6497-490d-afd8-dc95cbbf3c57","resolution":{"observed_at":"2026-08-06T20:24:27.822200Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:24:27.927688Z","title":"Swe-agent: Agent-computer interfaces enable automated software engineering","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.02825","last_updated":"2025-08-07T06:58:08Z","snapshot_observed_at":"2026-08-17T22:49:53.742436Z","submitted_at":"2025-07-03T17:35:31Z","title":"Establishing Best Practices for Building Rigorous Agentic Benchmarks","version":5},"reference_index":84,"source":"pdf_text","source_observed_at":"2026-08-06T20:24:27.927688Z"},"links":{"citing_paper":"/paper/2507.02825"},"observation_digest":"sha256:b60296714bba3e20151083528e5c99cd88928f403628ccfc680cff6397dde5e8","observation_id":"d4bb5d99-1118-437a-a708-e2affa5fc08c","resolution":{"observed_at":"2026-08-06T20:24:27.927688Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:24:27.984396Z","title":"React: Synergizing reasoning and acting in language models","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.02825","last_updated":"2025-08-07T06:58:08Z","snapshot_observed_at":"2026-08-17T22:49:53.742436Z","submitted_at":"2025-07-03T17:35:31Z","title":"Establishing Best Practices for Building Rigorous Agentic Benchmarks","version":5},"reference_index":85,"source":"pdf_text","source_observed_at":"2026-08-06T20:24:27.984396Z"},"links":{"citing_paper":"/paper/2507.02825"},"observation_digest":"sha256:62f47f07b0d06bd7b4b683b565ad2a269fd98bac2b26fc53142f4f9275e8d100","observation_id":"2c8fb028-6fbe-4b09-90f3-b6b4cf0f194d","resolution":{"observed_at":"2026-08-06T20:24:27.984396Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:24:35.929669Z","title":"tau-bench: A bench- mark for tool-agent-user interaction in real-world domains","venue":null,"work_id":"5f69a9a7-04d2-4ef9-bd96-ccccd183b219","year":2025},"citing_paper":{"arxiv_id":"2507.02825","last_updated":"2025-08-07T06:58:08Z","snapshot_observed_at":"2026-08-17T22:49:53.742436Z","submitted_at":"2025-07-03T17:35:31Z","title":"Establishing Best Practices for Building Rigorous Agentic Benchmarks","version":5},"reference_index":86,"source":"pdf_text","source_observed_at":"2026-08-06T20:24:28.051730Z"},"links":{"citing_paper":"/paper/2507.02825"},"observation_digest":"sha256:d5712101cc946864591a574c8d3146ec577c5ab7d8d151a445e1a025a526f68b","observation_id":"198fd22f-9441-476c-a031-85aa2f20fae5","resolution":{"observed_at":"2026-08-06T20:24:35.992119Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:24:35.756758Z","title":"Utboost: Rigorous evaluation of coding agents on swe-bench","venue":null,"work_id":"92d723a1-995a-4cda-b029-0a7a95d05748","year":2025},"citing_paper":{"arxiv_id":"2507.02825","last_updated":"2025-08-07T06:58:08Z","snapshot_observed_at":"2026-08-17T22:49:53.742436Z","submitted_at":"2025-07-03T17:35:31Z","title":"Establishing Best Practices for Building Rigorous Agentic Benchmarks","version":5},"reference_index":87,"source":"pdf_text","source_observed_at":"2026-08-06T20:24:28.132834Z"},"links":{"citing_paper":"/paper/2507.02825"},"observation_digest":"sha256:96fe650e2e0983be1a6320ef970fbf6ce0673eb8d7108eee31f699754688c993","observation_id":"d949f052-c124-4d97-82f6-19578d246cb3","resolution":{"observed_at":"2026-08-06T20:24:35.844195Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:24:35.628248Z","title":"Evaluating large language models at evaluating instruction following","venue":null,"work_id":"0c168ddd-6abc-4856-84f9-b7f1f4ecdab8","year":2024},"citing_paper":{"arxiv_id":"2507.02825","last_updated":"2025-08-07T06:58:08Z","snapshot_observed_at":"2026-08-17T22:49:53.742436Z","submitted_at":"2025-07-03T17:35:31Z","title":"Establishing Best Practices for Building Rigorous Agentic Benchmarks","version":5},"reference_index":88,"source":"pdf_text","source_observed_at":"2026-08-06T20:24:28.231761Z"},"links":{"citing_paper":"/paper/2507.02825"},"observation_digest":"sha256:fc6198f63e0823a1f3ed0c58c00e11ff39efc4637a9d5482f238e8571b0f9379","observation_id":"1a034863-6f59-44c1-a9f0-367e722d3b8b","resolution":{"observed_at":"2026-08-06T20:24:35.693140Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.08926","last_updated":"2025-04-12T21:26:07Z","snapshot_observed_at":"2026-08-20T14:10:03.882192Z","submitted_at":"2024-08-15T17:23:10Z","title":"Cybench: A Framework for Evaluating Cybersecurity Capabilities and Risks of Language Models","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2408.08926","snapshot_observed_at":"2026-08-06T20:24:28.287235Z","title":"Cybench: A framework for evaluating cybersecurity capabilities and risks of language models","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2507.02825","last_updated":"2025-08-07T06:58:08Z","snapshot_observed_at":"2026-08-17T22:49:53.742436Z","submitted_at":"2025-07-03T17:35:31Z","title":"Establishing Best Practices for Building Rigorous Agentic Benchmarks","version":5},"reference_index":89,"source":"pdf_text","source_observed_at":"2026-08-06T20:24:28.287235Z"},"links":{"cited_paper":"/paper/2408.08926","citing_paper":"/paper/2507.02825"},"observation_digest":"sha256:5aa8b5518a39645bb991ea2f0b096ad2b5309a4e207fa9280463e1b1bb6936f6","observation_id":"cb576ed1-53fe-4da0-ab9c-9c13e0aa1f7e","resolution":{"observed_at":"2026-08-06T20:24:28.287235Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:24:28.350185Z","title":"Judging llm-as-a-judge with mt-bench and chatbot arena","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.02825","last_updated":"2025-08-07T06:58:08Z","snapshot_observed_at":"2026-08-17T22:49:53.742436Z","submitted_at":"2025-07-03T17:35:31Z","title":"Establishing Best Practices for Building Rigorous Agentic Benchmarks","version":5},"reference_index":90,"source":"pdf_text","source_observed_at":"2026-08-06T20:24:28.350185Z"},"links":{"citing_paper":"/paper/2507.02825"},"observation_digest":"sha256:47aa5ee9b2005e34919a2dd7aa9b8e1753071f4b1bdb2b68ed722a5df3039770","observation_id":"684b05ef-1caf-4548-8578-8538f64cf57d","resolution":{"observed_at":"2026-08-06T20:24:28.350185Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2311.01964","last_updated":"2023-11-03T14:59:54Z","snapshot_observed_at":"2026-08-19T21:12:37.288621Z","submitted_at":"2023-11-03T14:59:54Z","title":"Don't Make Your LLM an Evaluation Benchmark Cheater","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2311.01964","snapshot_observed_at":"2026-08-06T20:24:28.438581Z","title":"Don’t make your llm an evaluation benchmark cheater","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.02825","last_updated":"2025-08-07T06:58:08Z","snapshot_observed_at":"2026-08-17T22:49:53.742436Z","submitted_at":"2025-07-03T17:35:31Z","title":"Establishing Best Practices for Building Rigorous Agentic Benchmarks","version":5},"reference_index":91,"source":"pdf_text","source_observed_at":"2026-08-06T20:24:28.438581Z"},"links":{"cited_paper":"/paper/2311.01964","citing_paper":"/paper/2507.02825"},"observation_digest":"sha256:b9c7eb2e51576de466cd0501ee58f29680d59bd2f874e059659c1eeb191e33c5","observation_id":"11652142-6a23-4b3c-b5d8-bee15e27fe51","resolution":{"observed_at":"2026-08-06T20:24:28.438581Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:24:35.462330Z","title":"X-webarena-leaderboard,","venue":null,"work_id":"94f671b1-0b9b-4547-91f1-70c501e92120","year":null},"citing_paper":{"arxiv_id":"2507.02825","last_updated":"2025-08-07T06:58:08Z","snapshot_observed_at":"2026-08-17T22:49:53.742436Z","submitted_at":"2025-07-03T17:35:31Z","title":"Establishing Best Practices for Building Rigorous Agentic Benchmarks","version":5},"reference_index":92,"source":"pdf_text","source_observed_at":"2026-08-06T20:24:28.512244Z"},"links":{"citing_paper":"/paper/2507.02825"},"observation_digest":"sha256:14f6f1554b4bb52c3d3d2dba7422e76e83c77fd17c92ea534696c30dced43c68","observation_id":"4171e63d-1c90-405f-bfc6-697c68ccd494","resolution":{"observed_at":"2026-08-06T20:24:35.549044Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:24:35.104641Z","title":"Webarena: A realistic web environment for build- ing autonomous agents","venue":null,"work_id":"9408c1d5-e26e-4024-a52a-7d576bda7577","year":2024},"citing_paper":{"arxiv_id":"2507.02825","last_updated":"2025-08-07T06:58:08Z","snapshot_observed_at":"2026-08-17T22:49:53.742436Z","submitted_at":"2025-07-03T17:35:31Z","title":"Establishing Best Practices for Building Rigorous Agentic Benchmarks","version":5},"reference_index":93,"source":"pdf_text","source_observed_at":"2026-08-06T20:24:28.701453Z"},"links":{"citing_paper":"/paper/2507.02825"},"observation_digest":"sha256:2c31505854450fee268ef6bc0433910a03353164f6beca540ced673e73e13ee1","observation_id":"ef5223c0-1a4a-4a77-9b95-a4220eb1bba5","resolution":{"observed_at":"2026-08-06T20:24:35.173406Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:24:28.775241Z","title":"Software unit test coverage and adequacy.Acm computing surveys (csur), 29(4):366–427, 1997","venue":null,"work_id":null,"year":1997},"citing_paper":{"arxiv_id":"2507.02825","last_updated":"2025-08-07T06:58:08Z","snapshot_observed_at":"2026-08-17T22:49:53.742436Z","submitted_at":"2025-07-03T17:35:31Z","title":"Establishing Best Practices for Building Rigorous Agentic Benchmarks","version":5},"reference_index":94,"source":"pdf_text","source_observed_at":"2026-08-06T20:24:28.775241Z"},"links":{"citing_paper":"/paper/2507.02825"},"observation_digest":"sha256:6e0af02e8f68db20695fca77aca3952e9495aaceb41d24ee9ab093f61ee21b59","observation_id":"2e34b126-d2fb-4384-ab41-5e41bc211521","resolution":{"observed_at":"2026-08-06T20:24:28.775241Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:24:34.922145Z","title":"Fuzzing: a survey for roadmap","venue":null,"work_id":"8af74ba4-31d5-407a-abf8-c474e83f7a0c","year":2022},"citing_paper":{"arxiv_id":"2507.02825","last_updated":"2025-08-07T06:58:08Z","snapshot_observed_at":"2026-08-17T22:49:53.742436Z","submitted_at":"2025-07-03T17:35:31Z","title":"Establishing Best Practices for Building Rigorous Agentic Benchmarks","version":5},"reference_index":95,"source":"pdf_text","source_observed_at":"2026-08-06T20:24:28.878325Z"},"links":{"citing_paper":"/paper/2507.02825"},"observation_digest":"sha256:90dedfab3ea05373c8724043f2d77133236d3b34548fd4d013c04b8e97783271","observation_id":"221e343c-f593-4261-8244-8f2cd2d7c753","resolution":{"observed_at":"2026-08-06T20:24:35.002208Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.17332","last_updated":"2025-06-24T04:10:59Z","snapshot_observed_at":"2026-08-20T05:27:00.273588Z","submitted_at":"2025-03-21T17:32:32Z","title":"CVE-Bench: A Benchmark for AI Agents' Ability to Exploit Real-World Web Application Vulnerabilities","version":4},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2503.17332","snapshot_observed_at":"2026-08-06T20:24:28.958503Z","title":"Cve-bench: A benchmark for ai agents’ ability to exploit real-world web application vulnerabilities","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.02825","last_updated":"2025-08-07T06:58:08Z","snapshot_observed_at":"2026-08-17T22:49:53.742436Z","submitted_at":"2025-07-03T17:35:31Z","title":"Establishing Best Practices for Building Rigorous Agentic Benchmarks","version":5},"reference_index":96,"source":"pdf_text","source_observed_at":"2026-08-06T20:24:28.958503Z"},"links":{"cited_paper":"/paper/2503.17332","citing_paper":"/paper/2507.02825"},"observation_digest":"sha256:686c56cdc17fffa36d4b29de56f25140233fddee9320bbd144f9d0da410ea24f","observation_id":"a67a9fdf-6975-4f4b-8497-98a52fbdd8aa","resolution":{"observed_at":"2026-08-06T20:24:28.958503Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.10934","last_updated":"2024-10-16T17:54:12Z","snapshot_observed_at":"2026-08-20T14:40:50.372602Z","submitted_at":"2024-10-14T17:57:02Z","title":"Agent-as-a-Judge: Evaluate Agents with Agents","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.10934","snapshot_observed_at":"2026-08-06T20:24:29.053233Z","title":"Agent- as-a-judge: Evaluate agents with agents","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.02825","last_updated":"2025-08-07T06:58:08Z","snapshot_observed_at":"2026-08-17T22:49:53.742436Z","submitted_at":"2025-07-03T17:35:31Z","title":"Establishing Best Practices for Building Rigorous Agentic Benchmarks","version":5},"reference_index":97,"source":"pdf_text","source_observed_at":"2026-08-06T20:24:29.053233Z"},"links":{"cited_paper":"/paper/2410.10934","citing_paper":"/paper/2507.02825"},"observation_digest":"sha256:9db19e9bdfd7efd75b8a621a4c449a395bb20e8b7ba3cffcb1fe84522f2ae7e9","observation_id":"2e183348-8063-425d-854b-dd8490835ce7","resolution":{"observed_at":"2026-08-06T20:24:29.053233Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:24:34.766576Z","title":"Can large language models transform computational social science? Computational Linguistics, 50 (1):237–291, 2024","venue":null,"work_id":"268d2e4b-5b0f-4c7b-b333-d199032f85d3","year":2024},"citing_paper":{"arxiv_id":"2507.02825","last_updated":"2025-08-07T06:58:08Z","snapshot_observed_at":"2026-08-17T22:49:53.742436Z","submitted_at":"2025-07-03T17:35:31Z","title":"Establishing Best Practices for Building Rigorous Agentic Benchmarks","version":5},"reference_index":98,"source":"pdf_text","source_observed_at":"2026-08-06T20:24:29.137040Z"},"links":{"citing_paper":"/paper/2507.02825"},"observation_digest":"sha256:3b881a1f72b9e04338f38c37e2ced92e33c5f916fc62c83fbd73b6c1d63bc271","observation_id":"f002e32b-3947-4bee-b6a8-e09f77291753","resolution":{"observed_at":"2026-08-06T20:24:34.833066Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:24:34.613437Z","title":null,"venue":null,"work_id":"051b6a55-2014-4fc7-968f-b68775386d35","year":null},"citing_paper":{"arxiv_id":"2507.02825","last_updated":"2025-08-07T06:58:08Z","snapshot_observed_at":"2026-08-17T22:49:53.742436Z","submitted_at":"2025-07-03T17:35:31Z","title":"Establishing Best Practices for Building Rigorous Agentic Benchmarks","version":5},"reference_index":100,"source":"pdf_text","source_observed_at":"2026-08-06T20:24:29.250524Z"},"links":{"citing_paper":"/paper/2507.02825"},"observation_digest":"sha256:e88efef8e3a9cd0fdfb6e5664a921a8c6ea3e2c28fe545467f706f628d86cc3c","observation_id":"8e8f9bf9-c634-4687-8e2a-6c2be41be038","resolution":{"observed_at":"2026-08-06T20:24:34.697054Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-06T20:24:34.449578Z","title":null,"venue":null,"work_id":"1f459b1c-4621-4e68-9651-a9f439ec1964","year":null},"citing_paper":{"arxiv_id":"2507.02825","last_updated":"2025-08-07T06:58:08Z","snapshot_observed_at":"2026-08-17T22:49:53.742436Z","submitted_at":"2025-07-03T17:35:31Z","title":"Establishing Best Practices for Building Rigorous Agentic Benchmarks","version":5},"reference_index":101,"source":"pdf_text","source_observed_at":"2026-08-06T20:24:29.338123Z"},"links":{"citing_paper":"/paper/2507.02825"},"observation_digest":"sha256:b3a37a7f57a52b474d32d8010155551a7fb3eb9166b9373d3107a09bf8813905","observation_id":"60fb9d5b-7a52-4cc3-9cd7-7be29b252f51","resolution":{"observed_at":"2026-08-06T20:24:34.520858Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-20T06:33:59.587034+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2507.02825","last_updated":"2025-08-07T06:58:08Z","latest_version":5,"primary_category":"cs.AI","snapshot_observed_at":"2026-08-17T22:49:53.742436Z","submitted_at":"2025-07-03T17:35:31Z","title":"Establishing Best Practices for Building Rigorous Agentic Benchmarks"},"reference_resolution":{"displayed":100,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":73,"verified_exact":0,"verified_fuzzy":27},"total_outbound_references":118},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-20T06:33:59.587034+00:00","source":"crossref"},{"observed_at":"2026-08-20T06:33:54.927442+00:00","source":"retraction_watch"}],"thesis":"As of 21 August 2026, this Paper Citation Record lists 100 of 118 outbound references and 37 inbound Pith citation observations for arXiv:2507.02825."}