{"as_of":"2026-08-07T20:54:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:4310e2e27a3b49c36ce7292b3899c72186b545ee20c124dcf2f78951ebaf7f91","coverage":[{"denominator":36,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":36,"source":"paper_references, paper_reference_links","source_observed_at":"2026-05-23T19:11:20.600633Z","state":"measured"},{"denominator":136,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":136,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-07T06:34:17.273281+00:00","state":"measured"},{"denominator":127,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":100,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T14:57:36.403917Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":1,"source":"pith","source_observed_at":"2026-08-05T02:28:24.338817Z","state":"measured"}],"external_citation_measurements":[{"count":9,"observed_at":"2026-08-05T02:28:24.338817Z","source":"pith"}],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2410.07095","last_updated":"2025-02-26T11:57:30Z","snapshot_observed_at":"2026-08-06T21:08:58.653035Z","submitted_at":"2024-10-09T17:34:27Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","version":6},"cited_work":{"arxiv_id":"2410.07095","doi":"10.48550/arxiv.2410.07095","metadata_source":"pith","pith_arxiv_id":"2410.07095","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","venue":"cs.CL","work_id":"a671e43f-ceab-49e7-adc3-473d802a97ca","year":2024},"citing_paper":{"arxiv_id":"2412.04984","last_updated":"2025-01-14T20:16:01Z","snapshot_observed_at":"2026-07-29T23:20:20.918596Z","submitted_at":"2024-12-06T12:09:50Z","title":"Frontier Models are Capable of In-context Scheming","version":2},"reference_index":10,"source":"arxiv_source","source_observed_at":"2026-05-16T14:22:01.616448Z"},"links":{"cited_paper":"/paper/2410.07095","citing_paper":"/paper/2412.04984"},"observation_digest":"sha256:ac53114118627edd4729d302fc93f1d2120fa0b3d3dee3df57c1aa66f52de3eb","observation_id":"0949488f-6b45-47ab-99f4-ccf2509dce71","resolution":{"observed_at":"2026-05-16T14:22:01.708833Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.07095","last_updated":"2025-02-26T11:57:30Z","snapshot_observed_at":"2026-08-06T21:08:58.653035Z","submitted_at":"2024-10-09T17:34:27Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","version":6},"cited_work":{"arxiv_id":"2410.07095","doi":"10.48550/arxiv.2410.07095","metadata_source":"pith","pith_arxiv_id":"2410.07095","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","venue":"cs.CL","work_id":"a671e43f-ceab-49e7-adc3-473d802a97ca","year":2024},"citing_paper":{"arxiv_id":"2501.14249","last_updated":"2026-02-20T04:23:01Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-01-24T05:27:46Z","title":"Humanity's Last Exam","version":10},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-05-10T18:40:50.139345Z"},"links":{"cited_paper":"/paper/2410.07095","citing_paper":"/paper/2501.14249"},"observation_digest":"sha256:2e43187a42efdd02afff07d8caedd56f7d48aacda783b1ef3708be13a4790121","observation_id":"2c8a8636-7558-4f7c-b51b-62ff7e988de6","resolution":{"observed_at":"2026-05-10T18:40:50.382147Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.07095","last_updated":"2025-02-26T11:57:30Z","snapshot_observed_at":"2026-08-06T21:08:58.653035Z","submitted_at":"2024-10-09T17:34:27Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","version":6},"cited_work":{"arxiv_id":"2410.07095","doi":"10.48550/arxiv.2410.07095","metadata_source":"pith","pith_arxiv_id":"2410.07095","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","venue":"cs.CL","work_id":"a671e43f-ceab-49e7-adc3-473d802a97ca","year":2024},"citing_paper":{"arxiv_id":"2503.09567","last_updated":"2025-07-18T15:57:54Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-03-12T17:35:03Z","title":"Towards Reasoning Era: A Survey of Long Chain-of-Thought for Reasoning Large Language Models","version":5},"reference_index":69,"source":"pdf_text","source_observed_at":"2026-05-12T08:40:40.910461Z"},"links":{"cited_paper":"/paper/2410.07095","citing_paper":"/paper/2503.09567"},"observation_digest":"sha256:81bff72dee696e34fc2b7b2e3b2ca337c5add8a9710e9948c77bd56e6e6b4baf","observation_id":"baf3ef74-fbc2-4445-8a92-07d2c91f88e8","resolution":{"observed_at":"2026-05-12T08:40:41.266922Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.07095","last_updated":"2025-02-26T11:57:30Z","snapshot_observed_at":"2026-08-06T21:08:58.653035Z","submitted_at":"2024-10-09T17:34:27Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","version":6},"cited_work":{"arxiv_id":"2410.07095","doi":"10.48550/arxiv.2410.07095","metadata_source":"pith","pith_arxiv_id":"2410.07095","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","venue":"cs.CL","work_id":"a671e43f-ceab-49e7-adc3-473d802a97ca","year":2024},"citing_paper":{"arxiv_id":"2503.21460","last_updated":"2025-03-27T12:50:17Z","snapshot_observed_at":"2026-07-06T20:59:35.694800Z","submitted_at":"2025-03-27T12:50:17Z","title":"Large Language Model Agent: A Survey on Methodology, Applications and Challenges","version":1},"reference_index":144,"source":"pdf_text","source_observed_at":"2026-05-22T21:51:34.309870Z"},"links":{"cited_paper":"/paper/2410.07095","citing_paper":"/paper/2503.21460"},"observation_digest":"sha256:df4949de2223ba5818cca91b99050b898978a42f60e732634ee1632e8056a5de","observation_id":"f883961b-a295-4de7-95c4-00bcd68091d6","resolution":{"observed_at":"2026-05-22T21:52:10.592130Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.07095","last_updated":"2025-02-26T11:57:30Z","snapshot_observed_at":"2026-08-06T21:08:58.653035Z","submitted_at":"2024-10-09T17:34:27Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","version":6},"cited_work":{"arxiv_id":"2410.07095","doi":"10.48550/arxiv.2410.07095","metadata_source":"pith","pith_arxiv_id":"2410.07095","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","venue":"cs.CL","work_id":"a671e43f-ceab-49e7-adc3-473d802a97ca","year":2024},"citing_paper":{"arxiv_id":"2504.19678","last_updated":"2026-03-06T19:01:27Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-04-28T11:08:22Z","title":"From LLM Reasoning to Autonomous AI Agents: A Comprehensive Review","version":2},"reference_index":127,"source":"pdf_text","source_observed_at":"2026-05-15T02:57:37.873567Z"},"links":{"cited_paper":"/paper/2410.07095","citing_paper":"/paper/2504.19678"},"observation_digest":"sha256:8d7f2938af5fcdac5bebd4c7b87304cbf5d9b45f3b7a90aeaa43ec4e720861fd","observation_id":"46a8763d-d797-452b-adba-009b1d51cbe2","resolution":{"observed_at":"2026-05-15T02:57:38.488809Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.07095","last_updated":"2025-02-26T11:57:30Z","snapshot_observed_at":"2026-08-06T21:08:58.653035Z","submitted_at":"2024-10-09T17:34:27Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.07095","snapshot_observed_at":"2026-08-07T14:57:36.403917Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.16832","last_updated":"2025-05-27T23:23:45Z","snapshot_observed_at":"2026-08-07T14:52:08.936212Z","submitted_at":"2025-05-22T16:02:18Z","title":"From EduVisBench to EduVisAgent: A Benchmark and Multi-Agent Framework for Reasoning-Driven Pedagogical Visualization","version":2},"reference_index":7,"source":"arxiv_source","source_observed_at":"2026-08-07T14:57:36.403917Z"},"links":{"cited_paper":"/paper/2410.07095","citing_paper":"/paper/2505.16832"},"observation_digest":"sha256:7b384bb1c5dee231fa9268828a5005e639b54af315e1c0f6e514e12e9fad513f","observation_id":"800e8d9e-d336-4dbf-b08d-e2ba52b7c0aa","resolution":{"observed_at":"2026-08-07T14:57:36.403917Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.07095","last_updated":"2025-02-26T11:57:30Z","snapshot_observed_at":"2026-08-06T21:08:58.653035Z","submitted_at":"2024-10-09T17:34:27Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.07095","snapshot_observed_at":"2026-08-07T14:11:51.566184Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2505.19683","last_updated":"2025-05-26T08:44:53Z","snapshot_observed_at":"2026-08-07T14:06:07.358887Z","submitted_at":"2025-05-26T08:44:53Z","title":"Large Language Models for Planning: A Comprehensive and Systematic Survey","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-07T14:11:51.566184Z"},"links":{"cited_paper":"/paper/2410.07095","citing_paper":"/paper/2505.19683"},"observation_digest":"sha256:83b82aff69790807e45555bf5e540c4709e2f54ebe96837145910dd8c533a3be","observation_id":"94911a56-4025-4428-a8db-5db3c2374852","resolution":{"observed_at":"2026-08-07T14:11:51.566184Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.07095","last_updated":"2025-02-26T11:57:30Z","snapshot_observed_at":"2026-08-06T21:08:58.653035Z","submitted_at":"2024-10-09T17:34:27Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.07095","snapshot_observed_at":"2026-08-07T13:53:16.130724Z","title":"Mle-bench: Evaluating machine learning agents on machine learning engineering","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.21577","last_updated":"2025-08-25T13:40:36Z","snapshot_observed_at":"2026-08-07T13:42:38.045978Z","submitted_at":"2025-05-27T08:35:05Z","title":"RepoMaster: Autonomous Exploration and Understanding of GitHub Repositories for Complex Task Solving","version":3},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-07T13:53:16.130724Z"},"links":{"cited_paper":"/paper/2410.07095","citing_paper":"/paper/2505.21577"},"observation_digest":"sha256:8b6ad26d652de0602f8355d0ea120dd4f3632df5b14639f88cba664c79533753","observation_id":"db9ab53c-6279-496e-9327-d61a9c33d19d","resolution":{"observed_at":"2026-08-07T13:53:16.130724Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.07095","last_updated":"2025-02-26T11:57:30Z","snapshot_observed_at":"2026-08-06T21:08:58.653035Z","submitted_at":"2024-10-09T17:34:27Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.07095","snapshot_observed_at":"2026-08-07T12:18:50.576064Z","title":"Mle-bench: Evaluating machine learning agents on machine learning engineering","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.24876","last_updated":"2026-05-24T03:59:43Z","snapshot_observed_at":"2026-08-07T12:10:13.725980Z","submitted_at":"2025-05-30T17:59:53Z","title":"Agent-X: Evaluating Deep Multimodal Reasoning in Vision-Centric Agentic Tasks","version":2},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-08-07T12:18:50.576064Z"},"links":{"cited_paper":"/paper/2410.07095","citing_paper":"/paper/2505.24876"},"observation_digest":"sha256:e7f91ee4fbe9ed13ec12c8d1b547a83f7e3176ea2cfafc66e6f623be4350193a","observation_id":"9cd5b506-3428-41fe-843e-60bc3e7b709a","resolution":{"observed_at":"2026-08-07T12:18:50.576064Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.07095","last_updated":"2025-02-26T11:57:30Z","snapshot_observed_at":"2026-08-06T21:08:58.653035Z","submitted_at":"2024-10-09T17:34:27Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.07095","snapshot_observed_at":"2026-08-07T11:49:01.526190Z","title":"Mle-bench: Evaluating machine learning agents on machine learning engineering.arXiv preprint arXiv:2410.07095,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.01372","last_updated":"2025-06-09T09:01:24Z","snapshot_observed_at":"2026-08-07T12:17:45.194954Z","submitted_at":"2025-06-02T06:59:10Z","title":"AI Scientists Fail Without Strong Implementation Capability","version":2},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-08-07T11:49:01.526190Z"},"links":{"cited_paper":"/paper/2410.07095","citing_paper":"/paper/2506.01372"},"observation_digest":"sha256:7ef28adfc196697c436805d00b44656eeb29fd610bf1e2b9ea5378c36c0e73ce","observation_id":"089f937e-d284-4d45-9032-4f160e40483b","resolution":{"observed_at":"2026-08-07T11:49:01.526190Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.07095","last_updated":"2025-02-26T11:57:30Z","snapshot_observed_at":"2026-08-06T21:08:58.653035Z","submitted_at":"2024-10-09T17:34:27Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","version":6},"cited_work":{"arxiv_id":"2410.07095","doi":"10.48550/arxiv.2410.07095","metadata_source":"pith","pith_arxiv_id":"2410.07095","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","venue":"cs.CL","work_id":"a671e43f-ceab-49e7-adc3-473d802a97ca","year":2024},"citing_paper":{"arxiv_id":"2506.02387","last_updated":"2026-04-13T08:26:12Z","snapshot_observed_at":"2026-08-04T07:16:14.991803Z","submitted_at":"2025-06-03T02:57:38Z","title":"VS-Bench: Evaluating VLMs for Strategic Abilities in Multi-Agent Environments","version":3},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-05-19T11:57:08.314088Z"},"links":{"cited_paper":"/paper/2410.07095","citing_paper":"/paper/2506.02387"},"observation_digest":"sha256:ea3129613ed8935450a0d4d55ace84c90dd3a646a87182a0abc65886d25d7f2d","observation_id":"2ed01d53-5a9b-436b-82be-455d36c104a1","resolution":{"observed_at":"2026-05-19T11:57:16.254409Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.07095","last_updated":"2025-02-26T11:57:30Z","snapshot_observed_at":"2026-08-06T21:08:58.653035Z","submitted_at":"2024-10-09T17:34:27Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.07095","snapshot_observed_at":"2026-08-07T10:51:57.932392Z","title":"S., Chowdhury, N., Jaffe, O., Aung, J., Sherburn, D., Mays, E., Starace, G., Liu, K., Maksin, L., Patwardhan, T., et al","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.04098","last_updated":"2025-06-10T13:14:14Z","snapshot_observed_at":"2026-08-07T16:06:45.124727Z","submitted_at":"2025-06-04T15:55:27Z","title":"TextAtari: 100K Frames Game Playing with Language Agents","version":2},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-07T10:51:57.932392Z"},"links":{"cited_paper":"/paper/2410.07095","citing_paper":"/paper/2506.04098"},"observation_digest":"sha256:45ac6566a9afdbcaca5c1d6b63eb229a65eee11f45e44d402f3fa6344914f14f","observation_id":"7765159c-2546-4784-b802-f379cc58522b","resolution":{"observed_at":"2026-08-07T10:51:57.932392Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.07095","last_updated":"2025-02-26T11:57:30Z","snapshot_observed_at":"2026-08-06T21:08:58.653035Z","submitted_at":"2024-10-09T17:34:27Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.07095","snapshot_observed_at":"2026-08-07T10:22:59.340621Z","title":"Mle-bench: Evaluating machine learning agents on machine learning engineering, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.05542","last_updated":"2025-06-05T19:44:38Z","snapshot_observed_at":"2026-08-07T10:17:06.625824Z","submitted_at":"2025-06-05T19:44:38Z","title":"Agentomics-ML: Autonomous Machine Learning Experimentation Agent for Genomic and Transcriptomic Data","version":1},"reference_index":4,"source":"arxiv_source","source_observed_at":"2026-08-07T10:22:59.340621Z"},"links":{"cited_paper":"/paper/2410.07095","citing_paper":"/paper/2506.05542"},"observation_digest":"sha256:f663e35d930a3ad098aecd76308e62294535dcea558491ac5f148de54f999ea7","observation_id":"00a96ab0-57aa-4f08-a269-a73cf4789e00","resolution":{"observed_at":"2026-08-07T10:22:59.340621Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.07095","last_updated":"2025-02-26T11:57:30Z","snapshot_observed_at":"2026-08-06T21:08:58.653035Z","submitted_at":"2024-10-09T17:34:27Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.07095","snapshot_observed_at":"2026-08-07T04:23:50.616857Z","title":"Mle-bench: Evaluating machine learning agents on real-world machine learning engineering tasks.arXiv preprint arXiv:2410.07095, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.10764","last_updated":"2025-06-12T14:46:41Z","snapshot_observed_at":"2026-08-07T11:34:37.934949Z","submitted_at":"2025-06-12T14:46:41Z","title":"OPT-BENCH: Evaluating LLM Agent on Large-Scale Search Spaces Optimization Problems","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-08-07T04:23:50.616857Z"},"links":{"cited_paper":"/paper/2410.07095","citing_paper":"/paper/2506.10764"},"observation_digest":"sha256:11ff0e124d74d2eede4644dbc7bdd9033f794a810544c588babac8fd95d68c43","observation_id":"b37b2185-8b21-4127-a86d-11fa9a8c23d5","resolution":{"observed_at":"2026-08-07T04:23:50.616857Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.07095","last_updated":"2025-02-26T11:57:30Z","snapshot_observed_at":"2026-08-06T21:08:58.653035Z","submitted_at":"2024-10-09T17:34:27Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.07095","snapshot_observed_at":"2026-08-07T06:00:19.820876Z","title":"Mle-bench: Eval- uating machine learning agents on machine learning engineering,","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.11102","last_updated":"2025-06-06T17:52:18Z","snapshot_observed_at":"2026-08-07T05:54:56.593167Z","submitted_at":"2025-06-06T17:52:18Z","title":"Evolutionary Perspectives on the Evaluation of LLM-Based AI Agents: A Comprehensive Survey","version":1},"reference_index":172,"source":"pdf_text","source_observed_at":"2026-08-07T06:00:19.820876Z"},"links":{"cited_paper":"/paper/2410.07095","citing_paper":"/paper/2506.11102"},"observation_digest":"sha256:d681dc2937170a7c0ee103960c973ec64493164b4c82ea91c45e1550d53b77d1","observation_id":"b9aa265e-c0c0-4265-b9a9-2354de16d98e","resolution":{"observed_at":"2026-08-07T06:00:19.820876Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.07095","last_updated":"2025-02-26T11:57:30Z","snapshot_observed_at":"2026-08-06T21:08:58.653035Z","submitted_at":"2024-10-09T17:34:27Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","version":6},"cited_work":{"arxiv_id":"2410.07095","doi":"10.48550/arxiv.2410.07095","metadata_source":"pith","pith_arxiv_id":"2410.07095","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","venue":"cs.CL","work_id":"a671e43f-ceab-49e7-adc3-473d802a97ca","year":2024},"citing_paper":{"arxiv_id":"2506.11763","last_updated":"2025-06-13T13:17:32Z","snapshot_observed_at":"2026-08-07T14:15:18.269910Z","submitted_at":"2025-06-13T13:17:32Z","title":"DeepResearch Bench: A Comprehensive Benchmark for Deep Research Agents","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-16T08:07:39.384613Z"},"links":{"cited_paper":"/paper/2410.07095","citing_paper":"/paper/2506.11763"},"observation_digest":"sha256:3ff4b2855505f80361cef49c6cce27d3ba19b761657b3c8a1508389db5f5b796","observation_id":"81445514-0b9c-4510-bfc1-5ad5d93c31d6","resolution":{"observed_at":"2026-05-16T08:07:39.426511Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.07095","last_updated":"2025-02-26T11:57:30Z","snapshot_observed_at":"2026-08-06T21:08:58.653035Z","submitted_at":"2024-10-09T17:34:27Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.07095","snapshot_observed_at":"2026-08-07T00:55:16.898887Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.12312","last_updated":"2025-06-14T02:22:28Z","snapshot_observed_at":"2026-08-07T12:23:07.508214Z","submitted_at":"2025-06-14T02:22:28Z","title":"Perspective on Utilizing Foundation Models for Laboratory Automation in Materials Research","version":1},"reference_index":63,"source":"pdf_text","source_observed_at":"2026-08-07T00:55:16.898887Z"},"links":{"cited_paper":"/paper/2410.07095","citing_paper":"/paper/2506.12312"},"observation_digest":"sha256:4b8d6cc881199dd897f14125a6f2fd031a40acd59fd42408e64a6e6178541ec0","observation_id":"1333024c-de16-4666-82d9-44d41a199367","resolution":{"observed_at":"2026-08-07T00:55:16.898887Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.07095","last_updated":"2025-02-26T11:57:30Z","snapshot_observed_at":"2026-08-06T21:08:58.653035Z","submitted_at":"2024-10-09T17:34:27Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","version":6},"cited_work":{"arxiv_id":"2410.07095","doi":"10.48550/arxiv.2410.07095","metadata_source":"pith","pith_arxiv_id":"2410.07095","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","venue":"cs.CL","work_id":"a671e43f-ceab-49e7-adc3-473d802a97ca","year":2024},"citing_paper":{"arxiv_id":"2506.22598","last_updated":"2026-04-21T20:17:15Z","snapshot_observed_at":"2026-08-03T23:13:09.503654Z","submitted_at":"2025-06-27T19:41:41Z","title":"RExBench: Can coding agents autonomously implement AI research extensions?","version":3},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-05-19T07:33:39.675929Z"},"links":{"cited_paper":"/paper/2410.07095","citing_paper":"/paper/2506.22598"},"observation_digest":"sha256:fd272f3302f30e05588983f22636929e6cc54e72baff8ae2cc99c47ed2f20d77","observation_id":"ccc444aa-7de8-44e9-b689-0b6714a020eb","resolution":{"observed_at":"2026-05-19T07:37:08.913299Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.07095","last_updated":"2025-02-26T11:57:30Z","snapshot_observed_at":"2026-08-06T21:08:58.653035Z","submitted_at":"2024-10-09T17:34:27Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.07095","snapshot_observed_at":"2026-08-06T21:37:50.650169Z","title":"Mle-bench: Evaluating machine learning agents on machine learning engineering","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.23719","last_updated":"2025-06-30T10:49:21Z","snapshot_observed_at":"2026-08-07T18:16:04.352755Z","submitted_at":"2025-06-30T10:49:21Z","title":"DABstep: Data Agent Benchmark for Multi-step Reasoning","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-06T21:37:50.650169Z"},"links":{"cited_paper":"/paper/2410.07095","citing_paper":"/paper/2506.23719"},"observation_digest":"sha256:8d1bd7382055fe00a86412a383ab5f8201a82b912d22aa88aaf9db6432155d70","observation_id":"dbeba95d-0dd7-46bf-9482-91eef5049960","resolution":{"observed_at":"2026-08-06T21:37:50.650169Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.07095","last_updated":"2025-02-26T11:57:30Z","snapshot_observed_at":"2026-08-06T21:08:58.653035Z","submitted_at":"2024-10-09T17:34:27Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.07095","snapshot_observed_at":"2026-08-06T20:45:12.210151Z","title":"Mle-bench: Evaluating machine learning agents on machine learning engineering.arXiv preprint arXiv:2410.07095, 2024","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.01903","last_updated":"2025-08-05T16:19:40Z","snapshot_observed_at":"2026-08-07T12:18:28.218519Z","submitted_at":"2025-07-02T17:19:20Z","title":"AI4Research: A Survey of Artificial Intelligence for Scientific Research","version":2},"reference_index":95,"source":"pdf_text","source_observed_at":"2026-08-06T20:45:12.210151Z"},"links":{"cited_paper":"/paper/2410.07095","citing_paper":"/paper/2507.01903"},"observation_digest":"sha256:da0a17388fa547262d971ac0ea3829242c1b2bba21ae7b23a57959ef4f415061","observation_id":"6fe5516d-1166-42c1-9628-e810171168ca","resolution":{"observed_at":"2026-08-06T20:45:12.210151Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.07095","last_updated":"2025-02-26T11:57:30Z","snapshot_observed_at":"2026-08-06T21:08:58.653035Z","submitted_at":"2024-10-09T17:34:27Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.07095","snapshot_observed_at":"2026-08-06T20:24:21.069258Z","title":"Mle-bench: Evaluating machine learning agents on machine learning engineering","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.02825","last_updated":"2025-08-07T06:58:08Z","snapshot_observed_at":"2026-08-06T22:15:42.048508Z","submitted_at":"2025-07-03T17:35:31Z","title":"Establishing Best Practices for Building Rigorous Agentic Benchmarks","version":5},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-08-06T20:24:21.069258Z"},"links":{"cited_paper":"/paper/2410.07095","citing_paper":"/paper/2507.02825"},"observation_digest":"sha256:6d213024a80ae6e7f0b9cf27c3b156eb899f60883392454de848ca849d811c15","observation_id":"c3080b94-bee8-43c4-a758-4a0b10a01473","resolution":{"observed_at":"2026-08-06T20:24:21.069258Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.07095","last_updated":"2025-02-26T11:57:30Z","snapshot_observed_at":"2026-08-06T21:08:58.653035Z","submitted_at":"2024-10-09T17:34:27Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.07095","snapshot_observed_at":"2026-08-06T19:00:08.554622Z","title":"org/CorpusID:260887105","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2507.08038","last_updated":"2026-06-01T08:32:26Z","snapshot_observed_at":"2026-08-06T18:52:28.453685Z","submitted_at":"2025-07-09T12:07:38Z","title":"AblationBench: Evaluating Automated Planning of Ablations in Empirical AI Research","version":3},"reference_index":2023,"source":"pdf_text","source_observed_at":"2026-08-06T19:00:08.554622Z"},"links":{"cited_paper":"/paper/2410.07095","citing_paper":"/paper/2507.08038"},"observation_digest":"sha256:0829c9cdb44ee1656bef5acc50f6205c6ec6919e7a437a0223dab2ca49d3516e","observation_id":"150e9c98-f8c7-42fb-922b-8887deee6850","resolution":{"observed_at":"2026-08-06T19:00:08.554622Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.07095","last_updated":"2025-02-26T11:57:30Z","snapshot_observed_at":"2026-08-06T21:08:58.653035Z","submitted_at":"2024-10-09T17:34:27Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.07095","snapshot_observed_at":"2026-08-06T18:12:29.955772Z","title":null,"venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.09089","last_updated":"2025-07-25T00:43:07Z","snapshot_observed_at":"2026-08-06T18:02:32.676640Z","submitted_at":"2025-07-12T00:16:33Z","title":"Measuring the Impact of Early-2025 AI on Experienced Open-Source Developer Productivity","version":2},"reference_index":2025,"source":"pdf_text","source_observed_at":"2026-08-06T18:12:29.955772Z"},"links":{"cited_paper":"/paper/2410.07095","citing_paper":"/paper/2507.09089"},"observation_digest":"sha256:fd2483e91390fb43b654ae415be3a50e8f45a2539fe8470ee1b65d79e69681cc","observation_id":"cc15c45f-a2ad-4758-adad-4896e7be1a57","resolution":{"observed_at":"2026-08-06T18:12:29.955772Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.07095","last_updated":"2025-02-26T11:57:30Z","snapshot_observed_at":"2026-08-06T21:08:58.653035Z","submitted_at":"2024-10-09T17:34:27Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","version":6},"cited_work":{"arxiv_id":"2410.07095","doi":"10.48550/arxiv.2410.07095","metadata_source":"pith","pith_arxiv_id":"2410.07095","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","venue":"cs.CL","work_id":"a671e43f-ceab-49e7-adc3-473d802a97ca","year":2024},"citing_paper":{"arxiv_id":"2507.21046","last_updated":"2026-01-16T20:59:08Z","snapshot_observed_at":"2026-08-01T06:32:44.461162Z","submitted_at":"2025-07-28T17:59:05Z","title":"A Survey of Self-Evolving Agents: What, When, How, and Where to Evolve on the Path to Artificial Super Intelligence","version":4},"reference_index":73,"source":"arxiv_source","source_observed_at":"2026-05-14T22:23:14.621091Z"},"links":{"cited_paper":"/paper/2410.07095","citing_paper":"/paper/2507.21046"},"observation_digest":"sha256:a5b415c0500226f7752a95698441a70104b565dbe430b393ca5dfb3fe1be0ecd","observation_id":"2e26fafb-125c-42f3-9fa6-98a8e1106f7f","resolution":{"observed_at":"2026-05-14T22:23:15.658139Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.07095","last_updated":"2025-02-26T11:57:30Z","snapshot_observed_at":"2026-08-06T21:08:58.653035Z","submitted_at":"2024-10-09T17:34:27Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.07095","snapshot_observed_at":"2026-08-06T10:55:14.616722Z","title":"Mle-bench: Evaluating machine learning agents on machine learning engineering","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.23276","last_updated":"2025-08-01T12:49:36Z","snapshot_observed_at":"2026-08-06T10:54:57.937285Z","submitted_at":"2025-07-31T06:32:06Z","title":"How Far Are AI Scientists from Changing the World?","version":2},"reference_index":24,"source":"arxiv_source","source_observed_at":"2026-08-06T10:55:14.616722Z"},"links":{"cited_paper":"/paper/2410.07095","citing_paper":"/paper/2507.23276"},"observation_digest":"sha256:58abff54ec97b3f306391e5a54432fd92c60388aff2fa9b306be0e806d715894","observation_id":"96bba6a7-7066-4c86-8529-f0aae9c18d72","resolution":{"observed_at":"2026-08-06T10:55:14.616722Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.07095","last_updated":"2025-02-26T11:57:30Z","snapshot_observed_at":"2026-08-06T21:08:58.653035Z","submitted_at":"2024-10-09T17:34:27Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.07095","snapshot_observed_at":"2026-08-06T10:32:33.646495Z","title":"Gemini Team","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2507.23701","last_updated":"2025-08-13T17:45:14Z","snapshot_observed_at":"2026-08-07T08:57:49.091605Z","submitted_at":"2025-07-31T16:22:55Z","title":"TextQuests: How Good are LLMs at Text-Based Video Games?","version":3},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-06T10:32:33.646495Z"},"links":{"cited_paper":"/paper/2410.07095","citing_paper":"/paper/2507.23701"},"observation_digest":"sha256:d0246123471d97cb550dacc49e838f3393eaf72b900301e26b6b0d9c983c5619","observation_id":"2c8caff6-8dcb-45c4-9c85-80b1ad948731","resolution":{"observed_at":"2026-08-06T10:32:33.646495Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.07095","last_updated":"2025-02-26T11:57:30Z","snapshot_observed_at":"2026-08-06T21:08:58.653035Z","submitted_at":"2024-10-09T17:34:27Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.07095","snapshot_observed_at":"2026-08-06T15:20:12.437386Z","title":"ICLR, 26 February","venue":null,"work_id":null,"year":2021},"citing_paper":{"arxiv_id":"2508.00875","last_updated":"2025-07-22T03:27:42Z","snapshot_observed_at":"2026-08-06T15:13:17.424058Z","submitted_at":"2025-07-22T03:27:42Z","title":"Preliminary suggestions for rigorous GPAI model evaluations","version":1},"reference_index":2025,"source":"pdf_text","source_observed_at":"2026-08-06T15:20:12.437386Z"},"links":{"cited_paper":"/paper/2410.07095","citing_paper":"/paper/2508.00875"},"observation_digest":"sha256:588b1b8a227adb9774942fa98506c9b251a02f9ab0a578cfd9fb0dd677c7993d","observation_id":"5b912a33-b4a6-4a11-bbfa-cfa0ac414b7e","resolution":{"observed_at":"2026-08-06T15:20:12.437386Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.07095","last_updated":"2025-02-26T11:57:30Z","snapshot_observed_at":"2026-08-06T21:08:58.653035Z","submitted_at":"2024-10-09T17:34:27Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","version":6},"cited_work":{"arxiv_id":"2410.07095","doi":"10.48550/arxiv.2410.07095","metadata_source":"pith","pith_arxiv_id":"2410.07095","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","venue":"cs.CL","work_id":"a671e43f-ceab-49e7-adc3-473d802a97ca","year":2024},"citing_paper":{"arxiv_id":"2508.10177","last_updated":"2026-04-22T18:28:23Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-08-13T20:29:56Z","title":"KompeteAI: Accelerated Autonomous Multi-Agent System for End-to-End Pipeline Generation for Machine Learning Problems","version":3},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-18T22:22:19.478156Z"},"links":{"cited_paper":"/paper/2410.07095","citing_paper":"/paper/2508.10177"},"observation_digest":"sha256:5661f28403c84ceb7157242e3a5875b25110a2367c9356d55b223159532ca0fa","observation_id":"46c82f76-f455-4970-8219-272ba1e8053b","resolution":{"observed_at":"2026-05-18T22:22:51.728367Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.07095","last_updated":"2025-02-26T11:57:30Z","snapshot_observed_at":"2026-08-06T21:08:58.653035Z","submitted_at":"2024-10-09T17:34:27Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.07095","snapshot_observed_at":"2026-08-05T15:53:47.697485Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2508.19461","last_updated":"2025-08-26T22:29:31Z","snapshot_observed_at":"2026-08-07T19:17:20.900756Z","submitted_at":"2025-08-26T22:29:31Z","title":"Reliable Weak-to-Strong Monitoring of LLM Agents","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-08-05T15:53:47.697485Z"},"links":{"cited_paper":"/paper/2410.07095","citing_paper":"/paper/2508.19461"},"observation_digest":"sha256:027fddd6d1c059a8dc561605dd287d610a68260769eba33561e6ec0dfc2a1965","observation_id":"e0f69ee3-1813-43c4-9ef2-efff9eb485ec","resolution":{"observed_at":"2026-08-05T15:53:47.697485Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.07095","last_updated":"2025-02-26T11:57:30Z","snapshot_observed_at":"2026-08-06T21:08:58.653035Z","submitted_at":"2024-10-09T17:34:27Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.07095","snapshot_observed_at":"2026-08-05T12:24:02.235163Z","title":"Mle-bench: Evaluating machine learning agents on machine learning engineering","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2509.01684","last_updated":"2025-09-01T18:04:10Z","snapshot_observed_at":"2026-08-05T12:24:00.598679Z","submitted_at":"2025-09-01T18:04:10Z","title":"Reinforcement Learning for Machine Learning Engineering Agents","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-05T12:24:02.235163Z"},"links":{"cited_paper":"/paper/2410.07095","citing_paper":"/paper/2509.01684"},"observation_digest":"sha256:9e813a59fb5eae5fa68888e2460bffae7d8701e7ba467d8a6e97a9adea0d548d","observation_id":"9966b5d1-224c-4fea-9a94-78f61bc54f1f","resolution":{"observed_at":"2026-08-05T12:24:02.235163Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.07095","last_updated":"2025-02-26T11:57:30Z","snapshot_observed_at":"2026-08-06T21:08:58.653035Z","submitted_at":"2024-10-09T17:34:27Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","version":6},"cited_work":{"arxiv_id":"2410.07095","doi":"10.48550/arxiv.2410.07095","metadata_source":"pith","pith_arxiv_id":"2410.07095","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","venue":"cs.CL","work_id":"a671e43f-ceab-49e7-adc3-473d802a97ca","year":2024},"citing_paper":{"arxiv_id":"2509.06806","last_updated":"2026-04-11T15:54:00Z","snapshot_observed_at":"2026-07-06T22:25:58.615083Z","submitted_at":"2025-09-08T15:38:31Z","title":"MachineLearningLM: Scaling Many-shot In-context Learning via Continued Pretraining","version":6},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-05-18T18:09:18.157131Z"},"links":{"cited_paper":"/paper/2410.07095","citing_paper":"/paper/2509.06806"},"observation_digest":"sha256:96e2e356d5b7dedb36d542f544596b3eeda629c94479986ce480bcc17fbb6d75","observation_id":"bc36038c-3468-43e8-a1e8-9d5666dab71a","resolution":{"observed_at":"2026-05-18T18:11:42.733862Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.07095","last_updated":"2025-02-26T11:57:30Z","snapshot_observed_at":"2026-08-06T21:08:58.653035Z","submitted_at":"2024-10-09T17:34:27Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","version":6},"cited_work":{"arxiv_id":"2410.07095","doi":"10.48550/arxiv.2410.07095","metadata_source":"pith","pith_arxiv_id":"2410.07095","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","venue":"cs.CL","work_id":"a671e43f-ceab-49e7-adc3-473d802a97ca","year":2024},"citing_paper":{"arxiv_id":"2509.08827","last_updated":"2025-10-09T17:08:52Z","snapshot_observed_at":"2026-08-06T15:38:05.011922Z","submitted_at":"2025-09-10T17:59:43Z","title":"A Survey of Reinforcement Learning for Large Reasoning Models","version":3},"reference_index":51,"source":"arxiv_source","source_observed_at":"2026-05-18T00:02:24.352947Z"},"links":{"cited_paper":"/paper/2410.07095","citing_paper":"/paper/2509.08827"},"observation_digest":"sha256:52486fdd869729d86259094aa95a2016dfad7b727d51efde6cf45a0595202281","observation_id":"66fb61dd-5ec4-455b-9983-47a6a7b6d04c","resolution":{"observed_at":"2026-05-18T00:02:25.003117Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.07095","last_updated":"2025-02-26T11:57:30Z","snapshot_observed_at":"2026-08-06T21:08:58.653035Z","submitted_at":"2024-10-09T17:34:27Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.07095","snapshot_observed_at":"2026-08-04T19:21:20.611481Z","title":"arXiv:2410.07095","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2509.09321","last_updated":"2025-09-11T10:10:48Z","snapshot_observed_at":"2026-08-07T16:36:25.604102Z","submitted_at":"2025-09-11T10:10:48Z","title":"Towards Adaptive ML Benchmarks: Web-Agent-Driven Construction, Domain Expansion, and Metric Optimization","version":1},"reference_index":2025,"source":"pdf_text","source_observed_at":"2026-08-04T19:21:20.611481Z"},"links":{"cited_paper":"/paper/2410.07095","citing_paper":"/paper/2509.09321"},"observation_digest":"sha256:41a11ea7c6309afac119a1eb83d916b5bbeaf9eaaa2f427280af4f5353abc69d","observation_id":"8efd4654-b1fb-4d81-bad9-179da1230ff3","resolution":{"observed_at":"2026-08-04T19:21:20.611481Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.07095","last_updated":"2025-02-26T11:57:30Z","snapshot_observed_at":"2026-08-06T21:08:58.653035Z","submitted_at":"2024-10-09T17:34:27Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","version":6},"cited_work":{"arxiv_id":"2410.07095","doi":"10.48550/arxiv.2410.07095","metadata_source":"pith","pith_arxiv_id":"2410.07095","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","venue":"cs.CL","work_id":"a671e43f-ceab-49e7-adc3-473d802a97ca","year":2024},"citing_paper":{"arxiv_id":"2510.17795","last_updated":"2026-04-19T16:34:29Z","snapshot_observed_at":"2026-07-06T22:33:39.148066Z","submitted_at":"2025-10-20T17:53:23Z","title":"What Makes AI Research Replicable? Executable Knowledge Graphs as Scientific Knowledge Representations","version":3},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-18T05:56:45.755148Z"},"links":{"cited_paper":"/paper/2410.07095","citing_paper":"/paper/2510.17795"},"observation_digest":"sha256:8d9c30a2a7c666feab7e73ae82406a5963fcae36560ade66d76e9d74c367435b","observation_id":"5c91b463-b855-4a13-a201-70374c10804e","resolution":{"observed_at":"2026-05-18T06:00:57.896540Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.07095","last_updated":"2025-02-26T11:57:30Z","snapshot_observed_at":"2026-08-06T21:08:58.653035Z","submitted_at":"2024-10-09T17:34:27Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.07095","snapshot_observed_at":"2026-08-04T08:59:45.080256Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2510.18003","last_updated":"2026-06-16T04:10:41Z","snapshot_observed_at":"2026-08-07T07:22:36.416388Z","submitted_at":"2025-10-20T18:37:11Z","title":"BadScientist: Can a Research Agent Write Convincing but Unsound Papers that Fool LLM Reviewers?","version":2},"reference_index":3,"source":"arxiv_source","source_observed_at":"2026-08-04T08:59:45.080256Z"},"links":{"cited_paper":"/paper/2410.07095","citing_paper":"/paper/2510.18003"},"observation_digest":"sha256:6f064aee603b962bb9eee52ec73f039f9c7422f6729b65814622529d548c1912","observation_id":"cdcbd144-bca0-4e94-9a72-46a5a6bc8518","resolution":{"observed_at":"2026-08-04T08:59:45.080256Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.07095","last_updated":"2025-02-26T11:57:30Z","snapshot_observed_at":"2026-08-06T21:08:58.653035Z","submitted_at":"2024-10-09T17:34:27Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","version":6},"cited_work":{"arxiv_id":"2410.07095","doi":"10.48550/arxiv.2410.07095","metadata_source":"pith","pith_arxiv_id":"2410.07095","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","venue":"cs.CL","work_id":"a671e43f-ceab-49e7-adc3-473d802a97ca","year":2024},"citing_paper":{"arxiv_id":"2512.09629","last_updated":"2026-05-08T15:34:03Z","snapshot_observed_at":"2026-07-06T22:38:35.387503Z","submitted_at":"2025-12-10T13:17:08Z","title":"End-to-end PDDL Planning with Hardcoded and Dynamic Agents","version":2},"reference_index":5,"source":"arxiv_source","source_observed_at":"2026-05-16T23:34:20.411940Z"},"links":{"cited_paper":"/paper/2410.07095","citing_paper":"/paper/2512.09629"},"observation_digest":"sha256:2c686975c8149580b2b24d0b57111582ef0561877bfd183022d2f21b4dabefe2","observation_id":"2290bce9-fdff-431d-9d56-67285c1d0f87","resolution":{"observed_at":"2026-05-16T23:38:41.675007Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.07095","last_updated":"2025-02-26T11:57:30Z","snapshot_observed_at":"2026-08-06T21:08:58.653035Z","submitted_at":"2024-10-09T17:34:27Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.07095","snapshot_observed_at":"2026-08-03T03:43:05.501625Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2602.07195","last_updated":"2026-07-27T21:39:01Z","snapshot_observed_at":"2026-08-03T03:43:04.755470Z","submitted_at":"2026-02-06T21:03:36Z","title":"Automated Modernization of Machine Learning Engineering Notebooks for Reproducibility","version":2},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-08-03T03:43:05.501625Z"},"links":{"cited_paper":"/paper/2410.07095","citing_paper":"/paper/2602.07195"},"observation_digest":"sha256:9217367e503b17642203b26efbb2f60015479e09200141f7df14a6edf4a9eab8","observation_id":"45a15d5e-76cb-48b3-8d58-1c69d11df068","resolution":{"observed_at":"2026-08-03T03:43:05.501625Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.07095","last_updated":"2025-02-26T11:57:30Z","snapshot_observed_at":"2026-08-06T21:08:58.653035Z","submitted_at":"2024-10-09T17:34:27Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","version":6},"cited_work":{"arxiv_id":"2410.07095","doi":"10.48550/arxiv.2410.07095","metadata_source":"pith","pith_arxiv_id":"2410.07095","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","venue":"cs.CL","work_id":"a671e43f-ceab-49e7-adc3-473d802a97ca","year":2024},"citing_paper":{"arxiv_id":"2602.07906","last_updated":"2026-05-07T09:44:38Z","snapshot_observed_at":"2026-07-06T22:45:00.816348Z","submitted_at":"2026-02-08T10:55:03Z","title":"AceGRPO: Adaptive Curriculum Enhanced Group Relative Policy Optimization for Autonomous Machine Learning Engineering","version":5},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-16T06:32:22.038300Z"},"links":{"cited_paper":"/paper/2410.07095","citing_paper":"/paper/2602.07906"},"observation_digest":"sha256:1a8668618ca0b620602ea4190be7e9ee88dc71bf19ad002fed6695f071a09b55","observation_id":"89386d3a-3e0e-46dc-aef9-00b813d0419a","resolution":{"observed_at":"2026-05-16T06:32:27.285326Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.07095","last_updated":"2025-02-26T11:57:30Z","snapshot_observed_at":"2026-08-06T21:08:58.653035Z","submitted_at":"2024-10-09T17:34:27Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.07095","snapshot_observed_at":"2026-08-02T23:25:33.220360Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2602.13937","last_updated":"2026-05-30T01:57:59Z","snapshot_observed_at":"2026-08-06T15:35:41.253186Z","submitted_at":"2026-02-15T00:20:58Z","title":"iML: Executable, Problem-Grounded, and Broadly Exploratory Code-Driven AutoML","version":2},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-02T23:25:33.220360Z"},"links":{"cited_paper":"/paper/2410.07095","citing_paper":"/paper/2602.13937"},"observation_digest":"sha256:634003df787995e4f39ef7738535d755fa6f3458c44cc47844d10687a1115c3f","observation_id":"73cfc36c-eea1-4951-b0be-8f9df34bd852","resolution":{"observed_at":"2026-08-02T23:25:33.220360Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.07095","last_updated":"2025-02-26T11:57:30Z","snapshot_observed_at":"2026-08-06T21:08:58.653035Z","submitted_at":"2024-10-09T17:34:27Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","version":6},"cited_work":{"arxiv_id":"2410.07095","doi":"10.48550/arxiv.2410.07095","metadata_source":"pith","pith_arxiv_id":"2410.07095","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","venue":"cs.CL","work_id":"a671e43f-ceab-49e7-adc3-473d802a97ca","year":2024},"citing_paper":{"arxiv_id":"2603.01692","last_updated":"2026-04-12T09:20:27Z","snapshot_observed_at":"2026-08-01T17:24:03.244671Z","submitted_at":"2026-03-02T10:22:47Z","title":"Reasoning as Gradient: Scaling MLE Agents Beyond Tree Search","version":3},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-15T17:49:46.383559Z"},"links":{"cited_paper":"/paper/2410.07095","citing_paper":"/paper/2603.01692"},"observation_digest":"sha256:e1c31f3c9bed52736ce9daafc0f912b5d4b3ea7799689da6bc614aeb8d322ac5","observation_id":"fc560ce5-11dd-48aa-a399-67b7b6147fd9","resolution":{"observed_at":"2026-05-15T17:50:12.253270Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.07095","last_updated":"2025-02-26T11:57:30Z","snapshot_observed_at":"2026-08-06T21:08:58.653035Z","submitted_at":"2024-10-09T17:34:27Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.07095","snapshot_observed_at":"2026-08-02T18:14:52.244321Z","title":"Mle-bench: Evaluating machine learning agents on machine learning engineering","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2603.14473","last_updated":"2026-07-15T10:22:14Z","snapshot_observed_at":"2026-08-05T20:09:51.962104Z","submitted_at":"2026-03-15T16:31:51Z","title":"AI Can Learn Scientific Taste","version":2},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-02T18:14:52.244321Z"},"links":{"cited_paper":"/paper/2410.07095","citing_paper":"/paper/2603.14473"},"observation_digest":"sha256:2431ba911be9ddc17fb432d4c55dc4b4609b3918aba73ac4aa6501376966dc53","observation_id":"4118b16a-0b55-44a3-a83a-f8d953c886b1","resolution":{"observed_at":"2026-08-02T18:14:52.244321Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.07095","last_updated":"2025-02-26T11:57:30Z","snapshot_observed_at":"2026-08-06T21:08:58.653035Z","submitted_at":"2024-10-09T17:34:27Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","version":6},"cited_work":{"arxiv_id":"2410.07095","doi":"10.48550/arxiv.2410.07095","metadata_source":"pith","pith_arxiv_id":"2410.07095","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","venue":"cs.CL","work_id":"a671e43f-ceab-49e7-adc3-473d802a97ca","year":2024},"citing_paper":{"arxiv_id":"2603.23964","last_updated":"2026-04-13T08:15:41Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2026-03-25T05:56:54Z","title":"From Pixels to Digital Agents: An Empirical Study on the Taxonomy and Technological Trends of Reinforcement Learning Environments","version":2},"reference_index":56,"source":"pdf_text","source_observed_at":"2026-05-15T01:20:03.181903Z"},"links":{"cited_paper":"/paper/2410.07095","citing_paper":"/paper/2603.23964"},"observation_digest":"sha256:a5f42668a5f822b81c6ed7258c9052ef244d4ca92fdf1e607d9a23fd4ce62462","observation_id":"b09ef985-2053-4c07-b40f-c57aa800b28c","resolution":{"observed_at":"2026-05-15T01:23:27.135290Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.07095","last_updated":"2025-02-26T11:57:30Z","snapshot_observed_at":"2026-08-06T21:08:58.653035Z","submitted_at":"2024-10-09T17:34:27Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","version":6},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2410.07095","snapshot_observed_at":"2026-07-13T16:50:50.552104Z","title":"arXiv preprint arXiv:2410.07095 (2024)","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2603.27742","last_updated":"2026-07-08T11:45:48Z","snapshot_observed_at":"2026-08-06T10:25:58.064750Z","submitted_at":"2026-03-29T15:50:36Z","title":"TIR-Agent: Training an Explorative and Efficient Agent for Image Restoration","version":2},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-07-13T16:50:50.552104Z"},"links":{"cited_paper":"/paper/2410.07095","citing_paper":"/paper/2603.27742"},"observation_digest":"sha256:03087355203d11f5c67199d676fbbf52c44c9219a5df1c19e1e024838e587829","observation_id":"2b64becb-1446-4d02-b3a9-1c144242b0f0","resolution":{"observed_at":"2026-07-13T16:50:50.552104Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2410.07095","last_updated":"2025-02-26T11:57:30Z","snapshot_observed_at":"2026-08-06T21:08:58.653035Z","submitted_at":"2024-10-09T17:34:27Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","version":6},"cited_work":{"arxiv_id":"2410.07095","doi":"10.48550/arxiv.2410.07095","metadata_source":"pith","pith_arxiv_id":"2410.07095","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","venue":"cs.CL","work_id":"a671e43f-ceab-49e7-adc3-473d802a97ca","year":2024},"citing_paper":{"arxiv_id":"2604.04532","last_updated":"2026-07-02T12:30:40Z","snapshot_observed_at":"2026-08-02T05:41:39.420925Z","submitted_at":"2026-04-06T08:54:16Z","title":"Multilingual Prompt Localization for Agent-as-a-Judge: Language and Backbone Sensitivity in Requirement-Level Evaluation","version":1},"reference_index":4,"source":"arxiv_source","source_observed_at":"2026-05-10T20:02:40.980538Z"},"links":{"cited_paper":"/paper/2410.07095","citing_paper":"/paper/2604.04532"},"observation_digest":"sha256:0f166f23921292a5057c5deabe193535acf29bc51ca085030c6a923f4afb8622","observation_id":"548d6f2c-9474-40b7-b595-4bdfdf214bb8","resolution":{"observed_at":"2026-05-10T22:15:50.744720Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.07095","last_updated":"2025-02-26T11:57:30Z","snapshot_observed_at":"2026-08-06T21:08:58.653035Z","submitted_at":"2024-10-09T17:34:27Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","version":6},"cited_work":{"arxiv_id":"2410.07095","doi":"10.48550/arxiv.2410.07095","metadata_source":"pith","pith_arxiv_id":"2410.07095","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","venue":"cs.CL","work_id":"a671e43f-ceab-49e7-adc3-473d802a97ca","year":2024},"citing_paper":{"arxiv_id":"2604.05912","last_updated":"2026-04-07T14:15:45Z","snapshot_observed_at":"2026-08-07T05:35:01.907926Z","submitted_at":"2026-04-07T14:15:45Z","title":"FrontierFinance: A Long-Horizon Computer-Use Benchmark of Real-World Financial Tasks","version":1},"reference_index":4,"source":"arxiv_source","source_observed_at":"2026-05-10T19:49:32.983778Z"},"links":{"cited_paper":"/paper/2410.07095","citing_paper":"/paper/2604.05912"},"observation_digest":"sha256:1ae268b68b2ab1904ba21386cf2a149e1de8135426e0c7c27c4237fdc32608d7","observation_id":"1485e507-b83d-48cb-9c8d-5c7be3e37f64","resolution":{"observed_at":"2026-05-10T22:30:49.944041Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.07095","last_updated":"2025-02-26T11:57:30Z","snapshot_observed_at":"2026-08-06T21:08:58.653035Z","submitted_at":"2024-10-09T17:34:27Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","version":6},"cited_work":{"arxiv_id":"2410.07095","doi":"10.48550/arxiv.2410.07095","metadata_source":"pith","pith_arxiv_id":"2410.07095","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","venue":"cs.CL","work_id":"a671e43f-ceab-49e7-adc3-473d802a97ca","year":2024},"citing_paper":{"arxiv_id":"2604.06169","last_updated":"2026-04-07T17:59:44Z","snapshot_observed_at":"2026-08-03T02:43:13.230039Z","submitted_at":"2026-04-07T17:59:44Z","title":"In-Place Test-Time Training","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-05-10T19:07:47.174513Z"},"links":{"cited_paper":"/paper/2410.07095","citing_paper":"/paper/2604.06169"},"observation_digest":"sha256:72a34e5655562b7bcbc48a9fceeee4ae7a7cae99fd3039460f2045f2ffe46699","observation_id":"5401f046-ac6e-40d6-9445-7369c63ebad1","resolution":{"observed_at":"2026-05-10T23:30:49.058898Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.07095","last_updated":"2025-02-26T11:57:30Z","snapshot_observed_at":"2026-08-06T21:08:58.653035Z","submitted_at":"2024-10-09T17:34:27Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","version":6},"cited_work":{"arxiv_id":"2410.07095","doi":"10.48550/arxiv.2410.07095","metadata_source":"pith","pith_arxiv_id":"2410.07095","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","venue":"cs.CL","work_id":"a671e43f-ceab-49e7-adc3-473d802a97ca","year":2024},"citing_paper":{"arxiv_id":"2604.09791","last_updated":"2026-04-10T18:13:09Z","snapshot_observed_at":"2026-07-06T22:58:34.504325Z","submitted_at":"2026-04-10T18:13:09Z","title":"Pioneer Agent: Continual Improvement of Small Language Models in Production","version":1},"reference_index":11,"source":"arxiv_source","source_observed_at":"2026-05-10T17:48:40.520740Z"},"links":{"cited_paper":"/paper/2410.07095","citing_paper":"/paper/2604.09791"},"observation_digest":"sha256:bff8165e6dac1fa7beeba86a16f75eb0eed55e4e937b97f4e7fbcdf85308910f","observation_id":"8c76dcaa-1c19-4cae-90fd-690547baf310","resolution":{"observed_at":"2026-05-11T06:05:57.568178Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.07095","last_updated":"2025-02-26T11:57:30Z","snapshot_observed_at":"2026-08-06T21:08:58.653035Z","submitted_at":"2024-10-09T17:34:27Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","version":6},"cited_work":{"arxiv_id":"2410.07095","doi":"10.48550/arxiv.2410.07095","metadata_source":"pith","pith_arxiv_id":"2410.07095","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","venue":"cs.CL","work_id":"a671e43f-ceab-49e7-adc3-473d802a97ca","year":2024},"citing_paper":{"arxiv_id":"2604.10718","last_updated":"2026-04-12T16:28:51Z","snapshot_observed_at":"2026-08-06T20:40:21.653968Z","submitted_at":"2026-04-12T16:28:51Z","title":"SciPredict: Can LLMs Predict the Outcomes of Scientific Experiments in Natural Sciences?","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-05-10T15:55:34.768853Z"},"links":{"cited_paper":"/paper/2410.07095","citing_paper":"/paper/2604.10718"},"observation_digest":"sha256:3a861b575327799131fb66a9df8512d53eef4d117a7b94ca1a2d903802d98a82","observation_id":"ea68e041-dfcd-410b-a9b8-b2602f0a0fb3","resolution":{"observed_at":"2026-05-11T09:36:03.503133Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.07095","last_updated":"2025-02-26T11:57:30Z","snapshot_observed_at":"2026-08-06T21:08:58.653035Z","submitted_at":"2024-10-09T17:34:27Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","version":6},"cited_work":{"arxiv_id":"2410.07095","doi":"10.48550/arxiv.2410.07095","metadata_source":"pith","pith_arxiv_id":"2410.07095","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","venue":"cs.CL","work_id":"a671e43f-ceab-49e7-adc3-473d802a97ca","year":2024},"citing_paper":{"arxiv_id":"2604.12102","last_updated":"2026-04-15T03:29:47Z","snapshot_observed_at":"2026-07-06T23:00:22.690154Z","submitted_at":"2026-04-13T22:22:07Z","title":"Spatial Atlas: Compute-Grounded Reasoning for Spatial-Aware Research Agent Benchmarks","version":2},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-10T15:28:03.449453Z"},"links":{"cited_paper":"/paper/2410.07095","citing_paper":"/paper/2604.12102"},"observation_digest":"sha256:2b85c9eac19ba2219feb8decd4e141d928b29f7df559f0920fc7802ba4592ed1","observation_id":"78afe47a-b5d9-48f7-b914-0cc5619ce32b","resolution":{"observed_at":"2026-05-11T10:31:01.230008Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.07095","last_updated":"2025-02-26T11:57:30Z","snapshot_observed_at":"2026-08-06T21:08:58.653035Z","submitted_at":"2024-10-09T17:34:27Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","version":6},"cited_work":{"arxiv_id":"2410.07095","doi":"10.48550/arxiv.2410.07095","metadata_source":"pith","pith_arxiv_id":"2410.07095","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","venue":"cs.CL","work_id":"a671e43f-ceab-49e7-adc3-473d802a97ca","year":2024},"citing_paper":{"arxiv_id":"2604.12290","last_updated":"2026-04-27T16:57:20Z","snapshot_observed_at":"2026-08-05T14:13:55.019169Z","submitted_at":"2026-04-14T05:02:06Z","title":"Frontier-Eng: Benchmarking Self-Evolving Agents on Real-World Engineering Tasks with Generative Optimization","version":2},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-10T16:17:32.290531Z"},"links":{"cited_paper":"/paper/2410.07095","citing_paper":"/paper/2604.12290"},"observation_digest":"sha256:17c0714c7a0bb5deaf615ca628e7c00caa2d0585a2b7b2da6c5bd979b8351598","observation_id":"64982a15-6358-4aa5-9359-19eef36708b0","resolution":{"observed_at":"2026-05-11T09:01:01.084341Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.07095","last_updated":"2025-02-26T11:57:30Z","snapshot_observed_at":"2026-08-06T21:08:58.653035Z","submitted_at":"2024-10-09T17:34:27Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","version":6},"cited_work":{"arxiv_id":"2410.07095","doi":"10.48550/arxiv.2410.07095","metadata_source":"pith","pith_arxiv_id":"2410.07095","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","venue":"cs.CL","work_id":"a671e43f-ceab-49e7-adc3-473d802a97ca","year":2024},"citing_paper":{"arxiv_id":"2604.14116","last_updated":"2026-04-22T07:33:13Z","snapshot_observed_at":"2026-07-06T23:02:00.082783Z","submitted_at":"2026-04-15T17:38:06Z","title":"TREX: Automating LLM Fine-tuning via Agent-Driven Tree-based Exploration","version":2},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-10T12:34:29.808503Z"},"links":{"cited_paper":"/paper/2410.07095","citing_paper":"/paper/2604.14116"},"observation_digest":"sha256:d0282ed888e2e17afe6d0bdf6c580544912137e623b32c0ff259ec6a78844864","observation_id":"680be77e-6b91-4632-bc70-5246ade11647","resolution":{"observed_at":"2026-05-11T11:46:34.784533Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.07095","last_updated":"2025-02-26T11:57:30Z","snapshot_observed_at":"2026-08-06T21:08:58.653035Z","submitted_at":"2024-10-09T17:34:27Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","version":6},"cited_work":{"arxiv_id":"2410.07095","doi":"10.48550/arxiv.2410.07095","metadata_source":"pith","pith_arxiv_id":"2410.07095","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","venue":"cs.CL","work_id":"a671e43f-ceab-49e7-adc3-473d802a97ca","year":2024},"citing_paper":{"arxiv_id":"2604.17406","last_updated":"2026-07-01T09:10:53Z","snapshot_observed_at":"2026-08-06T20:50:24.820204Z","submitted_at":"2026-04-19T12:26:05Z","title":"EvoMaster: A Foundational Evolving Agent Framework for Agentic Science at Scale","version":2},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-10T05:59:01.010437Z"},"links":{"cited_paper":"/paper/2410.07095","citing_paper":"/paper/2604.17406"},"observation_digest":"sha256:3f25aa8bb49138efe584a95d0bfdb56ee02ab5d7e0b4c142584bb3475095e095","observation_id":"a624fea8-8603-4239-beba-02a5a84879da","resolution":{"observed_at":"2026-05-10T06:01:13.394419Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.07095","last_updated":"2025-02-26T11:57:30Z","snapshot_observed_at":"2026-08-06T21:08:58.653035Z","submitted_at":"2024-10-09T17:34:27Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","version":6},"cited_work":{"arxiv_id":"2410.07095","doi":"10.48550/arxiv.2410.07095","metadata_source":"pith","pith_arxiv_id":"2410.07095","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","venue":"cs.CL","work_id":"a671e43f-ceab-49e7-adc3-473d802a97ca","year":2024},"citing_paper":{"arxiv_id":"2604.17406","last_updated":"2026-07-01T09:10:53Z","snapshot_observed_at":"2026-08-06T20:50:24.820204Z","submitted_at":"2026-04-19T12:26:05Z","title":"EvoMaster: A Foundational Evolving Agent Framework for Agentic Science at Scale","version":4},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-07-05T17:45:55.631459Z"},"links":{"cited_paper":"/paper/2410.07095","citing_paper":"/paper/2604.17406"},"observation_digest":"sha256:f78c36bcfab7333a40bde8201ad92b8a19485c39e6a753b1136b809e4199cd32","observation_id":"b08bc7e2-ff39-416d-9d0a-85ab7a1f7825","resolution":{"observed_at":"2026-07-05T17:51:14.765804Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.07095","last_updated":"2025-02-26T11:57:30Z","snapshot_observed_at":"2026-08-06T21:08:58.653035Z","submitted_at":"2024-10-09T17:34:27Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","version":6},"cited_work":{"arxiv_id":"2410.07095","doi":"10.48550/arxiv.2410.07095","metadata_source":"pith","pith_arxiv_id":"2410.07095","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","venue":"cs.CL","work_id":"a671e43f-ceab-49e7-adc3-473d802a97ca","year":2024},"citing_paper":{"arxiv_id":"2604.19341","last_updated":"2026-04-21T11:24:09Z","snapshot_observed_at":"2026-07-06T23:06:02.635464Z","submitted_at":"2026-04-21T11:24:09Z","title":"Evaluation-driven Scaling for Scientific Discovery","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-05-10T03:39:52.204043Z"},"links":{"cited_paper":"/paper/2410.07095","citing_paper":"/paper/2604.19341"},"observation_digest":"sha256:664fb1f0c499c672881286bfe8edd45f82138baac3175ac1e3f604b35bb2d706","observation_id":"e2136a57-a6db-45e2-8324-eec869a62ade","resolution":{"observed_at":"2026-05-11T12:26:07.565713Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.07095","last_updated":"2025-02-26T11:57:30Z","snapshot_observed_at":"2026-08-06T21:08:58.653035Z","submitted_at":"2024-10-09T17:34:27Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","version":6},"cited_work":{"arxiv_id":"2410.07095","doi":"10.48550/arxiv.2410.07095","metadata_source":"pith","pith_arxiv_id":"2410.07095","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","venue":"cs.CL","work_id":"a671e43f-ceab-49e7-adc3-473d802a97ca","year":2024},"citing_paper":{"arxiv_id":"2604.22230","last_updated":"2026-04-24T05:07:28Z","snapshot_observed_at":"2026-07-06T23:08:42.301487Z","submitted_at":"2026-04-24T05:07:28Z","title":"On Benchmark Hacking in ML Contests: Modeling, Insights and Design","version":1},"reference_index":9,"source":"arxiv_source","source_observed_at":"2026-05-08T09:18:34.040257Z"},"links":{"cited_paper":"/paper/2410.07095","citing_paper":"/paper/2604.22230"},"observation_digest":"sha256:7f32dc507e0787ccdaeab0ed151540022ddc1f3c8ecbdc928a67ee2cc8d0afad","observation_id":"80be0d3b-badd-4671-9d41-49cbf429c6c6","resolution":{"observed_at":"2026-05-11T20:21:12.738010Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.07095","last_updated":"2025-02-26T11:57:30Z","snapshot_observed_at":"2026-08-06T21:08:58.653035Z","submitted_at":"2024-10-09T17:34:27Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","version":6},"cited_work":{"arxiv_id":"2410.07095","doi":"10.48550/arxiv.2410.07095","metadata_source":"pith","pith_arxiv_id":"2410.07095","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","venue":"cs.CL","work_id":"a671e43f-ceab-49e7-adc3-473d802a97ca","year":2024},"citing_paper":{"arxiv_id":"2605.00347","last_updated":"2026-05-01T02:05:56Z","snapshot_observed_at":"2026-07-06T23:13:47.082550Z","submitted_at":"2026-05-01T02:05:56Z","title":"Odysseus: Scaling VLMs to 100+ Turn Decision-Making in Games via Reinforcement Learning","version":1},"reference_index":46,"source":"arxiv_source","source_observed_at":"2026-05-09T20:22:58.061772Z"},"links":{"cited_paper":"/paper/2410.07095","citing_paper":"/paper/2605.00347"},"observation_digest":"sha256:7f54523c285b1857e01392fb4b0b76c77f5c80d68bf869204c615cb2f4eaabdf","observation_id":"7cc74e0f-2b06-412a-a1ad-f53a219a33d8","resolution":{"observed_at":"2026-05-11T15:16:09.185864Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.07095","last_updated":"2025-02-26T11:57:30Z","snapshot_observed_at":"2026-08-06T21:08:58.653035Z","submitted_at":"2024-10-09T17:34:27Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","version":6},"cited_work":{"arxiv_id":"2410.07095","doi":"10.48550/arxiv.2410.07095","metadata_source":"pith","pith_arxiv_id":"2410.07095","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","venue":"cs.CL","work_id":"a671e43f-ceab-49e7-adc3-473d802a97ca","year":2024},"citing_paper":{"arxiv_id":"2605.02661","last_updated":"2026-05-04T14:40:42Z","snapshot_observed_at":"2026-07-06T23:15:41.793878Z","submitted_at":"2026-05-04T14:40:42Z","title":"AcademiClaw: When Students Set Challenges for AI Agents","version":1},"reference_index":2,"source":"arxiv_source","source_observed_at":"2026-05-08T19:24:29.696454Z"},"links":{"cited_paper":"/paper/2410.07095","citing_paper":"/paper/2605.02661"},"observation_digest":"sha256:2caba41bab64c0dd4c8e7f726fcfdba2b776fca0a26c245e368f6a85cc9ab638","observation_id":"fc6af751-d40f-4062-b432-34e55189aefd","resolution":{"observed_at":"2026-05-09T05:55:29.695068Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.07095","last_updated":"2025-02-26T11:57:30Z","snapshot_observed_at":"2026-08-06T21:08:58.653035Z","submitted_at":"2024-10-09T17:34:27Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","version":6},"cited_work":{"arxiv_id":"2410.07095","doi":"10.48550/arxiv.2410.07095","metadata_source":"pith","pith_arxiv_id":"2410.07095","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","venue":"cs.CL","work_id":"a671e43f-ceab-49e7-adc3-473d802a97ca","year":2024},"citing_paper":{"arxiv_id":"2605.07039","last_updated":"2026-05-07T23:38:50Z","snapshot_observed_at":"2026-08-02T16:36:23.528400Z","submitted_at":"2026-05-07T23:38:50Z","title":"PACEvolve++: Improving Test-time Learning for Evolutionary Search Agents","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-05-11T00:54:39.349292Z"},"links":{"cited_paper":"/paper/2410.07095","citing_paper":"/paper/2605.07039"},"observation_digest":"sha256:16035db9d6e4e50b53614483d468c9b4268a05d3748d8da014087910bf0c4e37","observation_id":"313fca62-7306-40cf-9175-65e1b277c37b","resolution":{"observed_at":"2026-05-11T05:00:56.387556Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.07095","last_updated":"2025-02-26T11:57:30Z","snapshot_observed_at":"2026-08-06T21:08:58.653035Z","submitted_at":"2024-10-09T17:34:27Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","version":6},"cited_work":{"arxiv_id":"2410.07095","doi":"10.48550/arxiv.2410.07095","metadata_source":"pith","pith_arxiv_id":"2410.07095","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","venue":"cs.CL","work_id":"a671e43f-ceab-49e7-adc3-473d802a97ca","year":2024},"citing_paper":{"arxiv_id":"2605.07073","last_updated":"2026-05-08T00:48:45Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2026-05-08T00:48:45Z","title":"TeamBench: Evaluating Agent Coordination under Enforced Role Separation","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-05-11T00:55:51.358828Z"},"links":{"cited_paper":"/paper/2410.07095","citing_paper":"/paper/2605.07073"},"observation_digest":"sha256:97d18bb9247fb2196410252ec835f08a43ce8c9753f34888460a04de226aeeb8","observation_id":"8f610c96-c7e3-462e-9038-b3956a66b57a","resolution":{"observed_at":"2026-05-11T05:00:55.525903Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.07095","last_updated":"2025-02-26T11:57:30Z","snapshot_observed_at":"2026-08-06T21:08:58.653035Z","submitted_at":"2024-10-09T17:34:27Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","version":6},"cited_work":{"arxiv_id":"2410.07095","doi":"10.48550/arxiv.2410.07095","metadata_source":"pith","pith_arxiv_id":"2410.07095","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","venue":"cs.CL","work_id":"a671e43f-ceab-49e7-adc3-473d802a97ca","year":2024},"citing_paper":{"arxiv_id":"2605.08904","last_updated":"2026-05-09T11:51:34Z","snapshot_observed_at":"2026-07-06T23:21:02.177557Z","submitted_at":"2026-05-09T11:51:34Z","title":"OPT-BENCH: Evaluating the Iterative Self-Optimization of LLM Agents in Large-Scale Search Spaces","version":1},"reference_index":142,"source":"arxiv_source","source_observed_at":"2026-05-12T02:57:15.521594Z"},"links":{"cited_paper":"/paper/2410.07095","citing_paper":"/paper/2605.08904"},"observation_digest":"sha256:70b498f46e783b6fb65065c855d81fab578888e3011905f26144456e625924b9","observation_id":"eb03d3b3-e7ad-4c7b-baa3-d79464aaf430","resolution":{"observed_at":"2026-05-12T03:01:18.592393Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.07095","last_updated":"2025-02-26T11:57:30Z","snapshot_observed_at":"2026-08-06T21:08:58.653035Z","submitted_at":"2024-10-09T17:34:27Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","version":6},"cited_work":{"arxiv_id":"2410.07095","doi":"10.48550/arxiv.2410.07095","metadata_source":"pith","pith_arxiv_id":"2410.07095","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","venue":"cs.CL","work_id":"a671e43f-ceab-49e7-adc3-473d802a97ca","year":2024},"citing_paper":{"arxiv_id":"2605.10906","last_updated":"2026-05-13T04:12:44Z","snapshot_observed_at":"2026-07-06T23:22:47.781940Z","submitted_at":"2026-05-11T17:46:24Z","title":"DataMaster: Data-Centric Autonomous AI Research","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-05-12T04:09:40.731228Z"},"links":{"cited_paper":"/paper/2410.07095","citing_paper":"/paper/2605.10906"},"observation_digest":"sha256:82a3b63ad60cc8fdabe5f6ccb6031b9ce2692de30d6d23119aa56a398da248e2","observation_id":"6e96672d-fb47-41b1-87d5-efd84d4ccf42","resolution":{"observed_at":"2026-05-12T06:36:24.801807Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.07095","last_updated":"2025-02-26T11:57:30Z","snapshot_observed_at":"2026-08-06T21:08:58.653035Z","submitted_at":"2024-10-09T17:34:27Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","version":6},"cited_work":{"arxiv_id":"2410.07095","doi":"10.48550/arxiv.2410.07095","metadata_source":"pith","pith_arxiv_id":"2410.07095","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","venue":"cs.CL","work_id":"a671e43f-ceab-49e7-adc3-473d802a97ca","year":2024},"citing_paper":{"arxiv_id":"2605.10906","last_updated":"2026-05-13T04:12:44Z","snapshot_observed_at":"2026-07-06T23:22:47.781940Z","submitted_at":"2026-05-11T17:46:24Z","title":"DataMaster: Data-Centric Autonomous AI Research","version":2},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-05-14T21:11:22.202161Z"},"links":{"cited_paper":"/paper/2410.07095","citing_paper":"/paper/2605.10906"},"observation_digest":"sha256:1cf3686087db609f21794d28bdfa8da237eb7a053515f3a8a6389c3c7202039c","observation_id":"f1c856d2-85f2-4075-a859-4985e291ed36","resolution":{"observed_at":"2026-05-14T21:12:59.133651Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.07095","last_updated":"2025-02-26T11:57:30Z","snapshot_observed_at":"2026-08-06T21:08:58.653035Z","submitted_at":"2024-10-09T17:34:27Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","version":6},"cited_work":{"arxiv_id":"2410.07095","doi":"10.48550/arxiv.2410.07095","metadata_source":"pith","pith_arxiv_id":"2410.07095","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","venue":"cs.CL","work_id":"a671e43f-ceab-49e7-adc3-473d802a97ca","year":2024},"citing_paper":{"arxiv_id":"2605.12015","last_updated":"2026-05-27T06:26:15Z","snapshot_observed_at":"2026-07-06T23:23:44.276236Z","submitted_at":"2026-05-12T12:03:54Z","title":"SkillSafetyBench: Evaluating Agent Safety under Skill-Facing Attack Surfaces","version":1},"reference_index":50,"source":"arxiv_source","source_observed_at":"2026-05-13T05:10:04.763149Z"},"links":{"cited_paper":"/paper/2410.07095","citing_paper":"/paper/2605.12015"},"observation_digest":"sha256:486cc04a6025edacc5e062cab77e33310b257a6539571148e58b2b292f68e58d","observation_id":"cfd3bf6c-cbdc-4ae8-b32b-7bda4deff2cd","resolution":{"observed_at":"2026-05-13T05:12:17.914966Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.07095","last_updated":"2025-02-26T11:57:30Z","snapshot_observed_at":"2026-08-06T21:08:58.653035Z","submitted_at":"2024-10-09T17:34:27Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","version":6},"cited_work":{"arxiv_id":"2410.07095","doi":"10.48550/arxiv.2410.07095","metadata_source":"pith","pith_arxiv_id":"2410.07095","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","venue":"cs.CL","work_id":"a671e43f-ceab-49e7-adc3-473d802a97ca","year":2024},"citing_paper":{"arxiv_id":"2605.12673","last_updated":"2026-05-12T19:22:45Z","snapshot_observed_at":"2026-08-02T02:31:32.394072Z","submitted_at":"2026-05-12T19:22:45Z","title":"Do Androids Dream of Breaking the Game? Systematically Auditing AI Agent Benchmarks with BenchJack","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-05-14T20:31:50.043920Z"},"links":{"cited_paper":"/paper/2410.07095","citing_paper":"/paper/2605.12673"},"observation_digest":"sha256:ddecf829d3d962a7343322d0f84899562f0d414206a1eb1c2678ca132b66e8da","observation_id":"f1f887bd-a17f-4a92-b2ef-065878966ef9","resolution":{"observed_at":"2026-05-14T20:32:56.909283Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.07095","last_updated":"2025-02-26T11:57:30Z","snapshot_observed_at":"2026-08-06T21:08:58.653035Z","submitted_at":"2024-10-09T17:34:27Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","version":6},"cited_work":{"arxiv_id":"2410.07095","doi":"10.48550/arxiv.2410.07095","metadata_source":"pith","pith_arxiv_id":"2410.07095","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","venue":"cs.CL","work_id":"a671e43f-ceab-49e7-adc3-473d802a97ca","year":2024},"citing_paper":{"arxiv_id":"2605.13634","last_updated":"2026-05-13T15:00:29Z","snapshot_observed_at":"2026-07-06T23:25:11.623026Z","submitted_at":"2026-05-13T15:00:29Z","title":"Europe and the Geopolitics of AGI: The Need for a Preparedness Plan","version":1},"reference_index":53,"source":"pdf_text","source_observed_at":"2026-05-14T17:43:38.233339Z"},"links":{"cited_paper":"/paper/2410.07095","citing_paper":"/paper/2605.13634"},"observation_digest":"sha256:90669f6bd07d76fe23c40a76fbcbb4e4ddbe294e10d6b761270fd0c2883088e4","observation_id":"4ba5177d-4501-49f5-b6e8-2acd5155a599","resolution":{"observed_at":"2026-05-14T17:49:23.579524Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.07095","last_updated":"2025-02-26T11:57:30Z","snapshot_observed_at":"2026-08-06T21:08:58.653035Z","submitted_at":"2024-10-09T17:34:27Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","version":6},"cited_work":{"arxiv_id":"2410.07095","doi":"10.48550/arxiv.2410.07095","metadata_source":"pith","pith_arxiv_id":"2410.07095","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","venue":"cs.CL","work_id":"a671e43f-ceab-49e7-adc3-473d802a97ca","year":2024},"citing_paper":{"arxiv_id":"2605.13950","last_updated":"2026-05-13T18:00:00Z","snapshot_observed_at":"2026-08-01T16:09:17.962725Z","submitted_at":"2026-05-13T18:00:00Z","title":"Collider-Bench: Benchmarking AI Agents with Particle Physics Analysis Reproduction","version":1},"reference_index":19,"source":"arxiv_source","source_observed_at":"2026-05-15T06:04:03.605898Z"},"links":{"cited_paper":"/paper/2410.07095","citing_paper":"/paper/2605.13950"},"observation_digest":"sha256:e1f268e9f955aa409bec0decc134a3f18a58b675c39076b2f2b2bb22cb3f74b5","observation_id":"df6e7b88-e280-4dcd-8151-e0557fa17296","resolution":{"observed_at":"2026-05-15T06:05:06.706952Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.07095","last_updated":"2025-02-26T11:57:30Z","snapshot_observed_at":"2026-08-06T21:08:58.653035Z","submitted_at":"2024-10-09T17:34:27Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","version":6},"cited_work":{"arxiv_id":"2410.07095","doi":"10.48550/arxiv.2410.07095","metadata_source":"pith","pith_arxiv_id":"2410.07095","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","venue":"cs.CL","work_id":"a671e43f-ceab-49e7-adc3-473d802a97ca","year":2024},"citing_paper":{"arxiv_id":"2605.14445","last_updated":"2026-05-14T06:39:42Z","snapshot_observed_at":"2026-08-01T15:27:40.908509Z","submitted_at":"2026-05-14T06:39:42Z","title":"FrontierSmith: Synthesizing Open-Ended Coding Problems at Scale","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-05-15T02:02:25.597640Z"},"links":{"cited_paper":"/paper/2410.07095","citing_paper":"/paper/2605.14445"},"observation_digest":"sha256:18440aca3eff3384d5b827e91d137a6b8b16db162e5b18a17c90f599085e718c","observation_id":"a239fdf7-7476-420a-abdd-f2561ed0e672","resolution":{"observed_at":"2026-05-15T02:03:28.882759Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.07095","last_updated":"2025-02-26T11:57:30Z","snapshot_observed_at":"2026-08-06T21:08:58.653035Z","submitted_at":"2024-10-09T17:34:27Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","version":6},"cited_work":{"arxiv_id":"2410.07095","doi":"10.48550/arxiv.2410.07095","metadata_source":"pith","pith_arxiv_id":"2410.07095","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","venue":"cs.CL","work_id":"a671e43f-ceab-49e7-adc3-473d802a97ca","year":2024},"citing_paper":{"arxiv_id":"2605.14790","last_updated":"2026-05-14T12:57:56Z","snapshot_observed_at":"2026-07-06T23:26:09.066863Z","submitted_at":"2026-05-14T12:57:56Z","title":"Graphs of Research: Citation Evolution Graphs as Supervision for Research Idea Generation","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-06-30T20:52:32.567107Z"},"links":{"cited_paper":"/paper/2410.07095","citing_paper":"/paper/2605.14790"},"observation_digest":"sha256:707c5ac2cb498ccde5aa490766dca6c035e23520574d2a9d87a034db02cc1e68","observation_id":"1516a089-4cdb-4397-a03d-9b2f3b7c0b19","resolution":{"observed_at":"2026-06-30T20:55:04.045232Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.07095","last_updated":"2025-02-26T11:57:30Z","snapshot_observed_at":"2026-08-06T21:08:58.653035Z","submitted_at":"2024-10-09T17:34:27Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","version":6},"cited_work":{"arxiv_id":"2410.07095","doi":"10.48550/arxiv.2410.07095","metadata_source":"pith","pith_arxiv_id":"2410.07095","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","venue":"cs.CL","work_id":"a671e43f-ceab-49e7-adc3-473d802a97ca","year":2024},"citing_paper":{"arxiv_id":"2605.15308","last_updated":"2026-05-14T18:21:08Z","snapshot_observed_at":"2026-08-02T20:42:03.805734Z","submitted_at":"2026-05-14T18:21:08Z","title":"SMCEvolve: Principled Scientific Discovery via Sequential Monte Carlo Evolution","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-05-19T16:22:12.940259Z"},"links":{"cited_paper":"/paper/2410.07095","citing_paper":"/paper/2605.15308"},"observation_digest":"sha256:f288c08bf117632f15bf9544585ab020f306103828335c089b37aff931d16698","observation_id":"97b19e26-c772-4343-9579-bc8081f1aac1","resolution":{"observed_at":"2026-05-19T16:22:39.499555Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.07095","last_updated":"2025-02-26T11:57:30Z","snapshot_observed_at":"2026-08-06T21:08:58.653035Z","submitted_at":"2024-10-09T17:34:27Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","version":6},"cited_work":{"arxiv_id":"2410.07095","doi":"10.48550/arxiv.2410.07095","metadata_source":"pith","pith_arxiv_id":"2410.07095","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","venue":"cs.CL","work_id":"a671e43f-ceab-49e7-adc3-473d802a97ca","year":2024},"citing_paper":{"arxiv_id":"2605.15766","last_updated":"2026-05-15T09:24:55Z","snapshot_observed_at":"2026-08-07T02:15:01.099029Z","submitted_at":"2026-05-15T09:24:55Z","title":"BioXArena: Benchmarking LLM Agents on Multi-Modal Biomedical Machine Learning Tasks","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-05-19T19:31:32.334837Z"},"links":{"cited_paper":"/paper/2410.07095","citing_paper":"/paper/2605.15766"},"observation_digest":"sha256:87cca825f1c76120b4507ddea5f7e9f205751ce14979a45cea7e9b61ed9c8429","observation_id":"a711806f-cefe-41e7-aff4-214ea632e467","resolution":{"observed_at":"2026-05-19T19:32:43.762882Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.07095","last_updated":"2025-02-26T11:57:30Z","snapshot_observed_at":"2026-08-06T21:08:58.653035Z","submitted_at":"2024-10-09T17:34:27Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","version":6},"cited_work":{"arxiv_id":"2410.07095","doi":"10.48550/arxiv.2410.07095","metadata_source":"pith","pith_arxiv_id":"2410.07095","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","venue":"cs.CL","work_id":"a671e43f-ceab-49e7-adc3-473d802a97ca","year":2024},"citing_paper":{"arxiv_id":"2605.16616","last_updated":"2026-05-15T20:35:32Z","snapshot_observed_at":"2026-07-06T23:27:43.762904Z","submitted_at":"2026-05-15T20:35:32Z","title":"MLReplicate: Benchmarking Autonomous Research Systems for Machine Learning Reproducibility","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-05-20T19:59:40.519962Z"},"links":{"cited_paper":"/paper/2410.07095","citing_paper":"/paper/2605.16616"},"observation_digest":"sha256:5541c559c31f2699fb20bb8ec59734a7b1513a68987f80c4e1c5418658c1456a","observation_id":"92bd12e4-4819-4baa-8d3b-88a8a9a2219f","resolution":{"observed_at":"2026-05-20T20:03:44.137430Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.07095","last_updated":"2025-02-26T11:57:30Z","snapshot_observed_at":"2026-08-06T21:08:58.653035Z","submitted_at":"2024-10-09T17:34:27Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","version":6},"cited_work":{"arxiv_id":"2410.07095","doi":"10.48550/arxiv.2410.07095","metadata_source":"pith","pith_arxiv_id":"2410.07095","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","venue":"cs.CL","work_id":"a671e43f-ceab-49e7-adc3-473d802a97ca","year":2024},"citing_paper":{"arxiv_id":"2605.17373","last_updated":"2026-05-29T04:24:32Z","snapshot_observed_at":"2026-07-06T23:28:21.395050Z","submitted_at":"2026-05-17T10:30:38Z","title":"FML-bench: A Controlled Study of AI Research Agent Strategies from the Perspective of Search Dynamics","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-05-20T14:25:15.565386Z"},"links":{"cited_paper":"/paper/2410.07095","citing_paper":"/paper/2605.17373"},"observation_digest":"sha256:f9fbd6f58457e981f325a0e4e49d0e7e61cd4a287f30de4dc4e493a73d0f3887","observation_id":"f29b7ab8-9d8a-4738-8d69-a85e6ac2b3f2","resolution":{"observed_at":"2026-05-20T14:28:21.432887Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.07095","last_updated":"2025-02-26T11:57:30Z","snapshot_observed_at":"2026-08-06T21:08:58.653035Z","submitted_at":"2024-10-09T17:34:27Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","version":6},"cited_work":{"arxiv_id":"2410.07095","doi":"10.48550/arxiv.2410.07095","metadata_source":"pith","pith_arxiv_id":"2410.07095","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","venue":"cs.CL","work_id":"a671e43f-ceab-49e7-adc3-473d802a97ca","year":2024},"citing_paper":{"arxiv_id":"2605.17373","last_updated":"2026-05-29T04:24:32Z","snapshot_observed_at":"2026-07-06T23:28:21.395050Z","submitted_at":"2026-05-17T10:30:38Z","title":"FML-bench: A Controlled Study of AI Research Agent Strategies from the Perspective of Search Dynamics","version":2},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-06-30T19:00:30.961402Z"},"links":{"cited_paper":"/paper/2410.07095","citing_paper":"/paper/2605.17373"},"observation_digest":"sha256:ec0049aaf738feceaf84ef4b3100f05e2c9e3ff82bacdefe04f94c60b983ef33","observation_id":"26dca1fb-2a0c-44d7-84b5-6d5f62d7a0c3","resolution":{"observed_at":"2026-06-30T19:05:00.903660Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.07095","last_updated":"2025-02-26T11:57:30Z","snapshot_observed_at":"2026-08-06T21:08:58.653035Z","submitted_at":"2024-10-09T17:34:27Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","version":6},"cited_work":{"arxiv_id":"2410.07095","doi":"10.48550/arxiv.2410.07095","metadata_source":"pith","pith_arxiv_id":"2410.07095","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","venue":"cs.CL","work_id":"a671e43f-ceab-49e7-adc3-473d802a97ca","year":2024},"citing_paper":{"arxiv_id":"2605.17439","last_updated":"2026-05-19T07:14:09Z","snapshot_observed_at":"2026-07-06T23:28:26.397778Z","submitted_at":"2026-05-17T13:22:22Z","title":"DiagEval: Trajectory-Conditioned Diagnosis for Reliable Software Evaluation with GUI Agents","version":1},"reference_index":33,"source":"arxiv_source","source_observed_at":"2026-05-19T23:09:56.130802Z"},"links":{"cited_paper":"/paper/2410.07095","citing_paper":"/paper/2605.17439"},"observation_digest":"sha256:9d5cb1899413345356dbceefe5ec78096d62da1968d24b7698c9f8ed64eb2d97","observation_id":"2d271753-d42e-4655-b891-26878c699197","resolution":{"observed_at":"2026-05-19T23:12:51.589025Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.07095","last_updated":"2025-02-26T11:57:30Z","snapshot_observed_at":"2026-08-06T21:08:58.653035Z","submitted_at":"2024-10-09T17:34:27Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","version":6},"cited_work":{"arxiv_id":"2410.07095","doi":"10.48550/arxiv.2410.07095","metadata_source":"pith","pith_arxiv_id":"2410.07095","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","venue":"cs.CL","work_id":"a671e43f-ceab-49e7-adc3-473d802a97ca","year":2024},"citing_paper":{"arxiv_id":"2605.17439","last_updated":"2026-05-19T07:14:09Z","snapshot_observed_at":"2026-07-06T23:28:26.397778Z","submitted_at":"2026-05-17T13:22:22Z","title":"DiagEval: Trajectory-Conditioned Diagnosis for Reliable Software Evaluation with GUI Agents","version":2},"reference_index":33,"source":"arxiv_source","source_observed_at":"2026-05-20T13:05:38.485058Z"},"links":{"cited_paper":"/paper/2410.07095","citing_paper":"/paper/2605.17439"},"observation_digest":"sha256:4e61bbdfd097e79cb650d2ad377e3638aff935dc0c6dc93c8d66a6e04079e676","observation_id":"ae64da6d-0fa5-465b-bcb9-14f40f62d585","resolution":{"observed_at":"2026-05-20T13:08:17.865759Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.07095","last_updated":"2025-02-26T11:57:30Z","snapshot_observed_at":"2026-08-06T21:08:58.653035Z","submitted_at":"2024-10-09T17:34:27Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","version":6},"cited_work":{"arxiv_id":"2410.07095","doi":"10.48550/arxiv.2410.07095","metadata_source":"pith","pith_arxiv_id":"2410.07095","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","venue":"cs.CL","work_id":"a671e43f-ceab-49e7-adc3-473d802a97ca","year":2024},"citing_paper":{"arxiv_id":"2605.17637","last_updated":"2026-05-21T20:02:32Z","snapshot_observed_at":"2026-08-03T12:30:05.317561Z","submitted_at":"2026-05-17T20:07:12Z","title":"WebGameBench: Requirement-to-Application Evaluation for Coding Agents via Browser-Native Games","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-05-20T12:24:06.062957Z"},"links":{"cited_paper":"/paper/2410.07095","citing_paper":"/paper/2605.17637"},"observation_digest":"sha256:9c67689ad8adde8461c5d86f44dc79444f0518e3661130aada25e5b9007c5cb1","observation_id":"47cadefa-3de7-42a4-8968-5415fc8a6fe2","resolution":{"observed_at":"2026-05-20T12:28:17.264615Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.07095","last_updated":"2025-02-26T11:57:30Z","snapshot_observed_at":"2026-08-06T21:08:58.653035Z","submitted_at":"2024-10-09T17:34:27Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","version":6},"cited_work":{"arxiv_id":"2410.07095","doi":"10.48550/arxiv.2410.07095","metadata_source":"pith","pith_arxiv_id":"2410.07095","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","venue":"cs.CL","work_id":"a671e43f-ceab-49e7-adc3-473d802a97ca","year":2024},"citing_paper":{"arxiv_id":"2605.17637","last_updated":"2026-05-21T20:02:32Z","snapshot_observed_at":"2026-08-03T12:30:05.317561Z","submitted_at":"2026-05-17T20:07:12Z","title":"WebGameBench: Requirement-to-Application Evaluation for Coding Agents via Browser-Native Games","version":2},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-05-25T05:45:04.573722Z"},"links":{"cited_paper":"/paper/2410.07095","citing_paper":"/paper/2605.17637"},"observation_digest":"sha256:42208392be9d941b0cbfb76987f3d0bb14d0612838b7e9e1cc342efa43d7c33e","observation_id":"00bc18d5-ec31-494c-8b6c-b2308cea62e8","resolution":{"observed_at":"2026-05-25T05:45:23.240288Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.07095","last_updated":"2025-02-26T11:57:30Z","snapshot_observed_at":"2026-08-06T21:08:58.653035Z","submitted_at":"2024-10-09T17:34:27Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","version":6},"cited_work":{"arxiv_id":"2410.07095","doi":"10.48550/arxiv.2410.07095","metadata_source":"pith","pith_arxiv_id":"2410.07095","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","venue":"cs.CL","work_id":"a671e43f-ceab-49e7-adc3-473d802a97ca","year":2024},"citing_paper":{"arxiv_id":"2605.19099","last_updated":"2026-05-18T20:37:14Z","snapshot_observed_at":"2026-07-31T05:54:44.970755Z","submitted_at":"2026-05-18T20:37:14Z","title":"DecisionBench: A Benchmark for Emergent Delegation in Long-Horizon Agentic Workflows","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-05-20T10:16:38.920528Z"},"links":{"cited_paper":"/paper/2410.07095","citing_paper":"/paper/2605.19099"},"observation_digest":"sha256:2b62122f4d5981e542154ee802c9abc715ce86c23d8075de606838e54b76b85d","observation_id":"f4293ff5-e60d-4736-9d3e-1c6fdbe8ad82","resolution":{"observed_at":"2026-05-20T10:18:11.752386Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.07095","last_updated":"2025-02-26T11:57:30Z","snapshot_observed_at":"2026-08-06T21:08:58.653035Z","submitted_at":"2024-10-09T17:34:27Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","version":6},"cited_work":{"arxiv_id":"2410.07095","doi":"10.48550/arxiv.2410.07095","metadata_source":"pith","pith_arxiv_id":"2410.07095","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","venue":"cs.CL","work_id":"a671e43f-ceab-49e7-adc3-473d802a97ca","year":2024},"citing_paper":{"arxiv_id":"2605.19156","last_updated":"2026-05-18T22:20:33Z","snapshot_observed_at":"2026-08-03T08:44:53.272172Z","submitted_at":"2026-05-18T22:20:33Z","title":"How Far Are We From True Auto-Research?","version":1},"reference_index":23,"source":"arxiv_source","source_observed_at":"2026-05-20T09:56:16.160551Z"},"links":{"cited_paper":"/paper/2410.07095","citing_paper":"/paper/2605.19156"},"observation_digest":"sha256:e64c037a62292d62469899e4acf32eefcfa047f65857359e81ca484ec094a957","observation_id":"5f4e81b0-4561-4a35-b96a-98a815f18d14","resolution":{"observed_at":"2026-05-20T09:58:11.265272Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.07095","last_updated":"2025-02-26T11:57:30Z","snapshot_observed_at":"2026-08-06T21:08:58.653035Z","submitted_at":"2024-10-09T17:34:27Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","version":6},"cited_work":{"arxiv_id":"2410.07095","doi":"10.48550/arxiv.2410.07095","metadata_source":"pith","pith_arxiv_id":"2410.07095","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","venue":"cs.CL","work_id":"a671e43f-ceab-49e7-adc3-473d802a97ca","year":2024},"citing_paper":{"arxiv_id":"2605.19447","last_updated":"2026-05-19T07:00:55Z","snapshot_observed_at":"2026-08-02T09:08:33.207894Z","submitted_at":"2026-05-19T07:00:55Z","title":"What and When to Distill: Selective Hindsight Distillation for Multi-Turn Agents","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-20T05:41:23.712146Z"},"links":{"cited_paper":"/paper/2410.07095","citing_paper":"/paper/2605.19447"},"observation_digest":"sha256:a27a815ac60a583628486707052f7b12bece34947a36a20cd25ec3d088e7e296","observation_id":"41b8d2f9-b593-42da-a0a8-1c5783be82f9","resolution":{"observed_at":"2026-05-20T05:43:05.771032Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.07095","last_updated":"2025-02-26T11:57:30Z","snapshot_observed_at":"2026-08-06T21:08:58.653035Z","submitted_at":"2024-10-09T17:34:27Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","version":6},"cited_work":{"arxiv_id":"2410.07095","doi":"10.48550/arxiv.2410.07095","metadata_source":"pith","pith_arxiv_id":"2410.07095","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","venue":"cs.CL","work_id":"a671e43f-ceab-49e7-adc3-473d802a97ca","year":2024},"citing_paper":{"arxiv_id":"2605.20025","last_updated":"2026-05-23T19:12:21Z","snapshot_observed_at":"2026-07-06T23:30:39.512029Z","submitted_at":"2026-05-19T15:49:51Z","title":"AutoResearchClaw: Self-Reinforcing Autonomous Research with Human-AI Collaboration","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-20T05:22:56.341333Z"},"links":{"cited_paper":"/paper/2410.07095","citing_paper":"/paper/2605.20025"},"observation_digest":"sha256:c19a7ca0f89c63009d5e1cbf6aa2fd3e3bf8cefeb06b93c6a79ce86bf198a5e2","observation_id":"02af144c-51c0-4476-bf9c-11c9f6a69597","resolution":{"observed_at":"2026-05-20T05:23:03.562193Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.07095","last_updated":"2025-02-26T11:57:30Z","snapshot_observed_at":"2026-08-06T21:08:58.653035Z","submitted_at":"2024-10-09T17:34:27Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","version":6},"cited_work":{"arxiv_id":"2410.07095","doi":"10.48550/arxiv.2410.07095","metadata_source":"pith","pith_arxiv_id":"2410.07095","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","venue":"cs.CL","work_id":"a671e43f-ceab-49e7-adc3-473d802a97ca","year":2024},"citing_paper":{"arxiv_id":"2605.20025","last_updated":"2026-05-23T19:12:21Z","snapshot_observed_at":"2026-07-06T23:30:39.512029Z","submitted_at":"2026-05-19T15:49:51Z","title":"AutoResearchClaw: Self-Reinforcing Autonomous Research with Human-AI Collaboration","version":2},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-06-30T18:07:57.905347Z"},"links":{"cited_paper":"/paper/2410.07095","citing_paper":"/paper/2605.20025"},"observation_digest":"sha256:8b7147f2e147fa7f1bb62339b88f7f5c338e7a33c281cf3c610acee7832c718e","observation_id":"1d4d4ea8-3b43-47c9-9d54-b10f82bdb35e","resolution":{"observed_at":"2026-07-01T15:05:47.884299Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.07095","last_updated":"2025-02-26T11:57:30Z","snapshot_observed_at":"2026-08-06T21:08:58.653035Z","submitted_at":"2024-10-09T17:34:27Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","version":6},"cited_work":{"arxiv_id":"2410.07095","doi":"10.48550/arxiv.2410.07095","metadata_source":"pith","pith_arxiv_id":"2410.07095","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","venue":"cs.CL","work_id":"a671e43f-ceab-49e7-adc3-473d802a97ca","year":2024},"citing_paper":{"arxiv_id":"2605.20086","last_updated":"2026-05-19T16:41:45Z","snapshot_observed_at":"2026-08-02T02:17:19.395211Z","submitted_at":"2026-05-19T16:41:45Z","title":"What Do Evolutionary Coding Agents Evolve?","version":1},"reference_index":59,"source":"pdf_text","source_observed_at":"2026-05-20T03:44:18.658541Z"},"links":{"cited_paper":"/paper/2410.07095","citing_paper":"/paper/2605.20086"},"observation_digest":"sha256:136539847ef4aa5b5433e5541508f6bc676d67de94cb1e467994bc88d232d988","observation_id":"e2ee0fa7-77a2-43b0-be19-d29bd853ec47","resolution":{"observed_at":"2026-05-20T03:48:03.327687Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.07095","last_updated":"2025-02-26T11:57:30Z","snapshot_observed_at":"2026-08-06T21:08:58.653035Z","submitted_at":"2024-10-09T17:34:27Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","version":6},"cited_work":{"arxiv_id":"2410.07095","doi":"10.48550/arxiv.2410.07095","metadata_source":"pith","pith_arxiv_id":"2410.07095","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","venue":"cs.CL","work_id":"a671e43f-ceab-49e7-adc3-473d802a97ca","year":2024},"citing_paper":{"arxiv_id":"2605.20306","last_updated":"2026-06-02T14:01:21Z","snapshot_observed_at":"2026-07-06T23:30:53.900592Z","submitted_at":"2026-05-19T15:08:34Z","title":"WildRoadBench: A Wild Aerial Road-Damage Grounding Benchmark for Vision-Language Models and Autonomous Agents","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-05-21T07:54:42.190296Z"},"links":{"cited_paper":"/paper/2410.07095","citing_paper":"/paper/2605.20306"},"observation_digest":"sha256:e79096e1e1338fe31a663d41ac847f9d111584a83e9628ef9418863f7b1b1bec","observation_id":"bbab3210-536e-4bbe-ae63-fdb65a033849","resolution":{"observed_at":"2026-05-21T07:54:49.600155Z","resolver_source":"local_arxiv","status":"malformed_identifier"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.07095","last_updated":"2025-02-26T11:57:30Z","snapshot_observed_at":"2026-08-06T21:08:58.653035Z","submitted_at":"2024-10-09T17:34:27Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","version":6},"cited_work":{"arxiv_id":"2410.07095","doi":"10.48550/arxiv.2410.07095","metadata_source":"pith","pith_arxiv_id":"2410.07095","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","venue":"cs.CL","work_id":"a671e43f-ceab-49e7-adc3-473d802a97ca","year":2024},"citing_paper":{"arxiv_id":"2605.20306","last_updated":"2026-06-02T14:01:21Z","snapshot_observed_at":"2026-07-06T23:30:53.900592Z","submitted_at":"2026-05-19T15:08:34Z","title":"WildRoadBench: A Wild Aerial Road-Damage Grounding Benchmark for Vision-Language Models and Autonomous Agents","version":2},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-06-30T18:07:22.464758Z"},"links":{"cited_paper":"/paper/2410.07095","citing_paper":"/paper/2605.20306"},"observation_digest":"sha256:e00c27a69c1ebc79191180a9381cbd73606be11eee48ebf8150ecd25576f6b63","observation_id":"104b269d-9077-47f1-80db-b816e75ff523","resolution":{"observed_at":"2026-06-30T18:14:59.143256Z","resolver_source":"local_arxiv","status":"malformed_identifier"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.07095","last_updated":"2025-02-26T11:57:30Z","snapshot_observed_at":"2026-08-06T21:08:58.653035Z","submitted_at":"2024-10-09T17:34:27Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","version":6},"cited_work":{"arxiv_id":"2410.07095","doi":"10.48550/arxiv.2410.07095","metadata_source":"pith","pith_arxiv_id":"2410.07095","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","venue":"cs.CL","work_id":"a671e43f-ceab-49e7-adc3-473d802a97ca","year":2024},"citing_paper":{"arxiv_id":"2605.20690","last_updated":"2026-05-26T04:52:11Z","snapshot_observed_at":"2026-07-06T23:31:13.042581Z","submitted_at":"2026-05-20T04:36:40Z","title":"Declarative Data Services: Structured Agentic Discovery for Composing Data Systems","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-05-21T05:16:46.549921Z"},"links":{"cited_paper":"/paper/2410.07095","citing_paper":"/paper/2605.20690"},"observation_digest":"sha256:0074b7ee33d87a55e0d3272c7b140780475a595f7605bf16692d215f9dc9c147","observation_id":"14b3caf7-8f5f-4193-9edf-63d6d36a4d44","resolution":{"observed_at":"2026-05-21T05:19:39.287742Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.07095","last_updated":"2025-02-26T11:57:30Z","snapshot_observed_at":"2026-08-06T21:08:58.653035Z","submitted_at":"2024-10-09T17:34:27Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","version":6},"cited_work":{"arxiv_id":"2410.07095","doi":"10.48550/arxiv.2410.07095","metadata_source":"pith","pith_arxiv_id":"2410.07095","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","venue":"cs.CL","work_id":"a671e43f-ceab-49e7-adc3-473d802a97ca","year":2024},"citing_paper":{"arxiv_id":"2605.20690","last_updated":"2026-05-26T04:52:11Z","snapshot_observed_at":"2026-07-06T23:31:13.042581Z","submitted_at":"2026-05-20T04:36:40Z","title":"Declarative Data Services: Structured Agentic Discovery for Composing Data Systems","version":2},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-06-30T17:46:34.881102Z"},"links":{"cited_paper":"/paper/2410.07095","citing_paper":"/paper/2605.20690"},"observation_digest":"sha256:630ab62c1f3ee8d863b6d5764dd4dda59c283c2dcc4a22334f406d55443358ef","observation_id":"7c85de97-2215-434f-8825-ee600e61f827","resolution":{"observed_at":"2026-07-01T15:05:48.125602Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.07095","last_updated":"2025-02-26T11:57:30Z","snapshot_observed_at":"2026-08-06T21:08:58.653035Z","submitted_at":"2024-10-09T17:34:27Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","version":6},"cited_work":{"arxiv_id":"2410.07095","doi":"10.48550/arxiv.2410.07095","metadata_source":"pith","pith_arxiv_id":"2410.07095","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","venue":"cs.CL","work_id":"a671e43f-ceab-49e7-adc3-473d802a97ca","year":2024},"citing_paper":{"arxiv_id":"2605.22154","last_updated":"2026-05-21T08:25:17Z","snapshot_observed_at":"2026-07-30T19:02:11.122375Z","submitted_at":"2026-05-21T08:25:17Z","title":"IdleSpec: Exploiting Idle Time via Speculative Planning for LLM Agents","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-05-22T06:21:56.971204Z"},"links":{"cited_paper":"/paper/2410.07095","citing_paper":"/paper/2605.22154"},"observation_digest":"sha256:613f4e5fd4e89ecdc86f9e0f0742234ed7f8df9cd4fc2697e92042c7d742727f","observation_id":"85350ddd-64b2-4a6c-b328-e22f820dc274","resolution":{"observed_at":"2026-05-22T06:24:40.483673Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.07095","last_updated":"2025-02-26T11:57:30Z","snapshot_observed_at":"2026-08-06T21:08:58.653035Z","submitted_at":"2024-10-09T17:34:27Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","version":6},"cited_work":{"arxiv_id":"2410.07095","doi":"10.48550/arxiv.2410.07095","metadata_source":"pith","pith_arxiv_id":"2410.07095","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","venue":"cs.CL","work_id":"a671e43f-ceab-49e7-adc3-473d802a97ca","year":2024},"citing_paper":{"arxiv_id":"2605.25045","last_updated":"2026-05-24T12:42:12Z","snapshot_observed_at":"2026-08-07T20:23:56.264448Z","submitted_at":"2026-05-24T12:42:12Z","title":"AION: Next-Generation Tasks and Practical Harness for Time Series","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-06-30T11:24:42.704735Z"},"links":{"cited_paper":"/paper/2410.07095","citing_paper":"/paper/2605.25045"},"observation_digest":"sha256:e59678688f9aaaf6d90ab8362f5daf0e05b4bc9cc2abf38319b87718504c967d","observation_id":"b4578c69-a663-4e2f-ae8e-ab40b82ddf6a","resolution":{"observed_at":"2026-06-30T11:54:38.596441Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.07095","last_updated":"2025-02-26T11:57:30Z","snapshot_observed_at":"2026-08-06T21:08:58.653035Z","submitted_at":"2024-10-09T17:34:27Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","version":6},"cited_work":{"arxiv_id":"2410.07095","doi":"10.48550/arxiv.2410.07095","metadata_source":"pith","pith_arxiv_id":"2410.07095","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","venue":"cs.CL","work_id":"a671e43f-ceab-49e7-adc3-473d802a97ca","year":2024},"citing_paper":{"arxiv_id":"2605.26340","last_updated":"2026-05-25T21:30:27Z","snapshot_observed_at":"2026-07-31T00:39:04.557291Z","submitted_at":"2026-05-25T21:30:27Z","title":"ScientistOne: Towards Human-Level Autonomous Research via Chain-of-Evidence","version":1},"reference_index":2,"source":"arxiv_source","source_observed_at":"2026-06-29T21:19:03.281629Z"},"links":{"cited_paper":"/paper/2410.07095","citing_paper":"/paper/2605.26340"},"observation_digest":"sha256:725e4d88677a12e6d8d9e4953eeebbb7fc12b482f8ba578d9dfe6892ee2573ab","observation_id":"1fc5fd31-db11-4b37-89bb-27d93630b493","resolution":{"observed_at":"2026-06-29T21:23:59.052296Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.07095","last_updated":"2025-02-26T11:57:30Z","snapshot_observed_at":"2026-08-06T21:08:58.653035Z","submitted_at":"2024-10-09T17:34:27Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","version":6},"cited_work":{"arxiv_id":"2410.07095","doi":"10.48550/arxiv.2410.07095","metadata_source":"pith","pith_arxiv_id":"2410.07095","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","venue":"cs.CL","work_id":"a671e43f-ceab-49e7-adc3-473d802a97ca","year":2024},"citing_paper":{"arxiv_id":"2605.30329","last_updated":"2026-05-28T17:57:37Z","snapshot_observed_at":"2026-08-06T08:03:02.263598Z","submitted_at":"2026-05-28T17:57:37Z","title":"SoundnessBench: Can Your AI Scientist Really Tell Good Research Ideas from Bad Ones?","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-06-29T08:13:42.770740Z"},"links":{"cited_paper":"/paper/2410.07095","citing_paper":"/paper/2605.30329"},"observation_digest":"sha256:a82a094e57686f5c540c0682233b7e4d9c4a6ed330f20f47451487429dd1b975","observation_id":"11e75caa-4fdc-43c9-8be3-57473bdf22ff","resolution":{"observed_at":"2026-06-29T08:23:15.805279Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.07095","last_updated":"2025-02-26T11:57:30Z","snapshot_observed_at":"2026-08-06T21:08:58.653035Z","submitted_at":"2024-10-09T17:34:27Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","version":6},"cited_work":{"arxiv_id":"2410.07095","doi":"10.48550/arxiv.2410.07095","metadata_source":"pith","pith_arxiv_id":"2410.07095","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","venue":"cs.CL","work_id":"a671e43f-ceab-49e7-adc3-473d802a97ca","year":2024},"citing_paper":{"arxiv_id":"2606.00051","last_updated":"2026-05-08T08:23:10Z","snapshot_observed_at":"2026-08-07T00:25:09.123345Z","submitted_at":"2026-05-08T08:23:10Z","title":"Business Utility of Large Language Models as Exploratory Data Analysis Agents","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-06-30T23:25:17.416071Z"},"links":{"cited_paper":"/paper/2410.07095","citing_paper":"/paper/2606.00051"},"observation_digest":"sha256:1d739741f5c810fe324a989a93208472b0f8f47a4b1d6904c811a2aebd2b39a0","observation_id":"d4cb0844-9fd9-46a9-a2ab-7445ea2a6490","resolution":{"observed_at":"2026-06-30T23:35:06.995731Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.07095","last_updated":"2025-02-26T11:57:30Z","snapshot_observed_at":"2026-08-06T21:08:58.653035Z","submitted_at":"2024-10-09T17:34:27Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","version":6},"cited_work":{"arxiv_id":"2410.07095","doi":"10.48550/arxiv.2410.07095","metadata_source":"pith","pith_arxiv_id":"2410.07095","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","venue":"cs.CL","work_id":"a671e43f-ceab-49e7-adc3-473d802a97ca","year":2024},"citing_paper":{"arxiv_id":"2606.00384","last_updated":"2026-06-07T19:48:24Z","snapshot_observed_at":"2026-07-06T23:41:05.637982Z","submitted_at":"2026-05-29T21:57:37Z","title":"VESTA: Visual Exploration with Statistical Tool Agents","version":2},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-06-28T21:58:11.339217Z"},"links":{"cited_paper":"/paper/2410.07095","citing_paper":"/paper/2606.00384"},"observation_digest":"sha256:16e969d21b1e3ccae63b469642e181877459d2225c40ae24e5dcaa2420a430ec","observation_id":"7c39c3e1-b733-4c05-a127-d018a7747b8f","resolution":{"observed_at":"2026-07-01T19:56:10.520983Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.07095","last_updated":"2025-02-26T11:57:30Z","snapshot_observed_at":"2026-08-06T21:08:58.653035Z","submitted_at":"2024-10-09T17:34:27Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","version":6},"cited_work":{"arxiv_id":"2410.07095","doi":"10.48550/arxiv.2410.07095","metadata_source":"pith","pith_arxiv_id":"2410.07095","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","venue":"cs.CL","work_id":"a671e43f-ceab-49e7-adc3-473d802a97ca","year":2024},"citing_paper":{"arxiv_id":"2606.01961","last_updated":"2026-06-03T05:43:27Z","snapshot_observed_at":"2026-08-07T15:01:13.401479Z","submitted_at":"2026-06-01T09:22:55Z","title":"AutoMedBench: Towards Medical AutoResearch with Agentic AI Models","version":2},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-06-28T14:38:40.017263Z"},"links":{"cited_paper":"/paper/2410.07095","citing_paper":"/paper/2606.01961"},"observation_digest":"sha256:0cd05be3497ee83194f98db5635ebe1f0afcffc6f717a7797f23a3e40ef21097","observation_id":"ac48592d-2c15-40fc-9674-b48ffeda9bf4","resolution":{"observed_at":"2026-07-01T23:06:21.108710Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.07095","last_updated":"2025-02-26T11:57:30Z","snapshot_observed_at":"2026-08-06T21:08:58.653035Z","submitted_at":"2024-10-09T17:34:27Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","version":6},"cited_work":{"arxiv_id":"2410.07095","doi":"10.48550/arxiv.2410.07095","metadata_source":"pith","pith_arxiv_id":"2410.07095","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","venue":"cs.CL","work_id":"a671e43f-ceab-49e7-adc3-473d802a97ca","year":2024},"citing_paper":{"arxiv_id":"2606.04261","last_updated":"2026-06-02T22:26:53Z","snapshot_observed_at":"2026-08-02T20:22:03.575430Z","submitted_at":"2026-06-02T22:26:53Z","title":"Can Generalist Agents Automate Data Curation?","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-06-28T09:32:39.361415Z"},"links":{"cited_paper":"/paper/2410.07095","citing_paper":"/paper/2606.04261"},"observation_digest":"sha256:e4d1eadacc4f2c6ed07166f99262faa0f3e7f9730cfdb88320eb44f7a99f37e5","observation_id":"ba8327f8-3fcf-45f8-97e5-32533a365df6","resolution":{"observed_at":"2026-07-02T03:56:35.137529Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.07095","last_updated":"2025-02-26T11:57:30Z","snapshot_observed_at":"2026-08-06T21:08:58.653035Z","submitted_at":"2024-10-09T17:34:27Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","version":6},"cited_work":{"arxiv_id":"2410.07095","doi":"10.48550/arxiv.2410.07095","metadata_source":"pith","pith_arxiv_id":"2410.07095","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","venue":"cs.CL","work_id":"a671e43f-ceab-49e7-adc3-473d802a97ca","year":2024},"citing_paper":{"arxiv_id":"2606.04455","last_updated":"2026-06-03T04:58:17Z","snapshot_observed_at":"2026-08-06T11:33:07.154821Z","submitted_at":"2026-06-03T04:58:17Z","title":"The Meta-Agent Challenge: Are Current Agents Capable of Autonomous Agent Development?","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-06-28T06:29:42.665765Z"},"links":{"cited_paper":"/paper/2410.07095","citing_paper":"/paper/2606.04455"},"observation_digest":"sha256:93a82ed79dd1cac5b2d2068c5f8beb094d4475f170dfd6fca315f303fdc269d2","observation_id":"bb561dac-0119-4a07-b03e-764941fb9271","resolution":{"observed_at":"2026-07-02T07:56:47.678461Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.07095","last_updated":"2025-02-26T11:57:30Z","snapshot_observed_at":"2026-08-06T21:08:58.653035Z","submitted_at":"2024-10-09T17:34:27Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","version":6},"cited_work":{"arxiv_id":"2410.07095","doi":"10.48550/arxiv.2410.07095","metadata_source":"pith","pith_arxiv_id":"2410.07095","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","venue":"cs.CL","work_id":"a671e43f-ceab-49e7-adc3-473d802a97ca","year":2024},"citing_paper":{"arxiv_id":"2606.05250","last_updated":"2026-06-03T12:56:11Z","snapshot_observed_at":"2026-07-06T23:45:19.055840Z","submitted_at":"2026-06-03T12:56:11Z","title":"Towards Persistent Case-Based Memory for Autonomous Data Science: A CBR-Augmented R&D-Agent with a Locally Deployable Small Language Model","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-06-28T05:19:14.975753Z"},"links":{"cited_paper":"/paper/2410.07095","citing_paper":"/paper/2606.05250"},"observation_digest":"sha256:888fbe41c3ce01cbd98889d78add1ed511539c15f2e3a8bcc6c7cfc85308eedf","observation_id":"c2edd2e1-602e-4a3b-aa4e-74a69caf9de2","resolution":{"observed_at":"2026-07-02T10:06:51.727212Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.07095","last_updated":"2025-02-26T11:57:30Z","snapshot_observed_at":"2026-08-06T21:08:58.653035Z","submitted_at":"2024-10-09T17:34:27Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","version":6},"cited_work":{"arxiv_id":"2410.07095","doi":"10.48550/arxiv.2410.07095","metadata_source":"pith","pith_arxiv_id":"2410.07095","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","venue":"cs.CL","work_id":"a671e43f-ceab-49e7-adc3-473d802a97ca","year":2024},"citing_paper":{"arxiv_id":"2606.11176","last_updated":"2026-06-09T17:51:55Z","snapshot_observed_at":"2026-08-04T10:02:13.493460Z","submitted_at":"2026-06-09T17:51:55Z","title":"Data Journalist Agent: Transforming Data into Verifiable Multimodal Stories","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-06-27T13:43:02.919248Z"},"links":{"cited_paper":"/paper/2410.07095","citing_paper":"/paper/2606.11176"},"observation_digest":"sha256:d89c104b621646029091a89f5015bacfb5ee5f7acf57314535b87350920d739c","observation_id":"53dc74d2-7240-4d29-8666-bc7c511b3a01","resolution":{"observed_at":"2026-07-03T04:37:37.769214Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.07095","last_updated":"2025-02-26T11:57:30Z","snapshot_observed_at":"2026-08-06T21:08:58.653035Z","submitted_at":"2024-10-09T17:34:27Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","version":6},"cited_work":{"arxiv_id":"2410.07095","doi":"10.48550/arxiv.2410.07095","metadata_source":"pith","pith_arxiv_id":"2410.07095","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","venue":"cs.CL","work_id":"a671e43f-ceab-49e7-adc3-473d802a97ca","year":2024},"citing_paper":{"arxiv_id":"2606.11522","last_updated":"2026-06-09T23:55:31Z","snapshot_observed_at":"2026-08-07T10:05:03.526098Z","submitted_at":"2026-06-09T23:55:31Z","title":"Search Discipline for Long-Horizon Research Agents","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-06-27T12:51:07.758929Z"},"links":{"cited_paper":"/paper/2410.07095","citing_paper":"/paper/2606.11522"},"observation_digest":"sha256:e65b3febe8d0653e5a9ee119f9e1b35fa74b8e39efd62e85723c13b8cbadc88f","observation_id":"63a7b378-6de1-4e7d-8e33-4a76c8fd6d57","resolution":{"observed_at":"2026-06-27T13:30:57.161480Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.07095","last_updated":"2025-02-26T11:57:30Z","snapshot_observed_at":"2026-08-06T21:08:58.653035Z","submitted_at":"2024-10-09T17:34:27Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","version":6},"cited_work":{"arxiv_id":"2410.07095","doi":"10.48550/arxiv.2410.07095","metadata_source":"pith","pith_arxiv_id":"2410.07095","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","venue":"cs.CL","work_id":"a671e43f-ceab-49e7-adc3-473d802a97ca","year":2024},"citing_paper":{"arxiv_id":"2606.11926","last_updated":"2026-06-10T10:57:05Z","snapshot_observed_at":"2026-07-06T23:50:53.469137Z","submitted_at":"2026-06-10T10:57:05Z","title":"Toward Generalist Autonomous Research via Hypothesis-Tree Refinement","version":1},"reference_index":115,"source":"arxiv_source","source_observed_at":"2026-06-27T09:34:41.800309Z"},"links":{"cited_paper":"/paper/2410.07095","citing_paper":"/paper/2606.11926"},"observation_digest":"sha256:6bd99a2653a391b1ada494424e68fa81728e18c3ba02b64072d555c97a5154c7","observation_id":"63ca5120-9179-411c-96a6-ee7e58da9f14","resolution":{"observed_at":"2026-06-27T09:40:47.023998Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2410.07095/citation-record","integrity":"/paper/2410.07095/integrity","json":"/paper/2410.07095/citation-record.json","paper":"/paper/2410.07095"},"outbound":[{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Anthropic's Responsible Scaling Policy , Version 1.0, September 2023","venue":null,"work_id":"28183d11-a7c9-4af0-accc-29b0eb8518b6","year":2023},"citing_paper":{"arxiv_id":"2410.07095","last_updated":"2025-02-26T11:57:30Z","snapshot_observed_at":"2026-08-06T21:08:58.653035Z","submitted_at":"2024-10-09T17:34:27Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","version":6},"reference_index":1,"source":"arxiv_source","source_observed_at":"2026-05-23T19:11:20.600633Z"},"links":{"citing_paper":"/paper/2410.07095"},"observation_digest":"sha256:7e095d424ffb891ba44955cf1ea27152f11f60b453408ac756f449b98b88323c","observation_id":"10058669-c66c-4d10-931e-d0de89239b7b","resolution":{"observed_at":"2026-05-23T20:03:26.251461Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2108.07732","last_updated":"2021-08-16T03:57:30Z","snapshot_observed_at":"2026-08-02T19:23:53.535075Z","submitted_at":"2021-08-16T03:57:30Z","title":"Program Synthesis with Large Language Models","version":1},"cited_work":{"arxiv_id":"2108.07732","doi":"10.1007/s11390-025-5518-5","metadata_source":"pith","pith_arxiv_id":"2108.07732","snapshot_observed_at":"2026-07-11T11:50:26.030339Z","title":"Program Synthesis with Large Language Models","venue":"cs.PL","work_id":"fd241a05-03b9-4de2-9588-9d77ce176125","year":2021},"citing_paper":{"arxiv_id":"2410.07095","last_updated":"2025-02-26T11:57:30Z","snapshot_observed_at":"2026-08-06T21:08:58.653035Z","submitted_at":"2024-10-09T17:34:27Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","version":6},"reference_index":2,"source":"arxiv_source","source_observed_at":"2026-05-23T19:11:20.600633Z"},"links":{"cited_paper":"/paper/2108.07732","citing_paper":"/paper/2410.07095"},"observation_digest":"sha256:d70dfe622a32191d2d4f14dc94430e5400821898e31962aadd1bab6429d0133c","observation_id":"b41e221d-f8cb-4ff6-9116-74d91283c89c","resolution":{"observed_at":"2026-05-23T19:13:21.578703Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2202.07646","last_updated":"2023-03-06T06:28:18Z","snapshot_observed_at":"2026-08-03T19:51:05.627063Z","submitted_at":"2022-02-15T18:48:31Z","title":"Quantifying Memorization Across Neural Language Models","version":3},"cited_work":{"arxiv_id":"2202.07646","doi":"10.48550/arxiv.2202.07646","metadata_source":"pith","pith_arxiv_id":"2202.07646","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Quantifying Memorization Across Neural Language Models","venue":"cs.LG","work_id":"35487ec1-b90b-4ace-95bd-1bce30064b2e","year":2022},"citing_paper":{"arxiv_id":"2410.07095","last_updated":"2025-02-26T11:57:30Z","snapshot_observed_at":"2026-08-06T21:08:58.653035Z","submitted_at":"2024-10-09T17:34:27Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","version":6},"reference_index":3,"source":"arxiv_source","source_observed_at":"2026-05-23T19:11:20.600633Z"},"links":{"cited_paper":"/paper/2202.07646","citing_paper":"/paper/2410.07095"},"observation_digest":"sha256:054507496ad565c906d4bce9be9238284cdb2d384a0e08c47cc9ebfc9a31fb1e","observation_id":"69eb685f-89f0-40b6-97f7-cdb6cb72e6bd","resolution":{"observed_at":"2026-05-23T19:13:21.574142Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2107.03374","last_updated":"2021-07-14T17:16:02Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2021-07-07T17:41:24Z","title":"Evaluating Large Language Models Trained on Code","version":2},"cited_work":{"arxiv_id":"2107.03374","doi":"10.48550/arxiv.2107.03374","metadata_source":"pith","pith_arxiv_id":"2107.03374","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Evaluating Large Language Models Trained on Code","venue":"cs.LG","work_id":"042493e9-b26f-4b4e-bbde-382072ca9b08","year":2021},"citing_paper":{"arxiv_id":"2410.07095","last_updated":"2025-02-26T11:57:30Z","snapshot_observed_at":"2026-08-06T21:08:58.653035Z","submitted_at":"2024-10-09T17:34:27Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","version":6},"reference_index":4,"source":"arxiv_source","source_observed_at":"2026-05-23T19:11:20.600633Z"},"links":{"cited_paper":"/paper/2107.03374","citing_paper":"/paper/2410.07095"},"observation_digest":"sha256:d97ab56585ec136fe08c7e4693c8a28f2f3159d24a03d5e1bf9391b021479b4f","observation_id":"31abeee7-f11b-4692-9173-46311c65a0b1","resolution":{"observed_at":"2026-05-23T19:13:21.570148Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-01T08:08:23.404839+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-01T08:08:23.404839+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Cognition Introducing Devin , the first AI software engineer, March 2024","venue":null,"work_id":"b52a8fb2-d012-41d0-8320-d4da00dd9554","year":2024},"citing_paper":{"arxiv_id":"2410.07095","last_updated":"2025-02-26T11:57:30Z","snapshot_observed_at":"2026-08-06T21:08:58.653035Z","submitted_at":"2024-10-09T17:34:27Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","version":6},"reference_index":5,"source":"arxiv_source","source_observed_at":"2026-05-23T19:11:20.600633Z"},"links":{"citing_paper":"/paper/2410.07095"},"observation_digest":"sha256:600497336054f961d8a3b5aaba58f5e60ac0d0c8f2becc9851f93d4fab2a3d31","observation_id":"a2285312-d1d7-4a8c-b93e-ec90e7c1b685","resolution":{"observed_at":"2026-05-23T20:03:26.247456Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Openvaccine: Covid-19 mrna vaccine degradation prediction","venue":null,"work_id":"28676f9c-c99c-46e1-893b-31e880b94f64","year":2020},"citing_paper":{"arxiv_id":"2410.07095","last_updated":"2025-02-26T11:57:30Z","snapshot_observed_at":"2026-08-06T21:08:58.653035Z","submitted_at":"2024-10-09T17:34:27Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","version":6},"reference_index":6,"source":"arxiv_source","source_observed_at":"2026-05-23T19:11:20.600633Z"},"links":{"citing_paper":"/paper/2410.07095"},"observation_digest":"sha256:6b67950ea987550a71d233093ff80fc55b1ad7ac9113cf7038b5fffcdde6f937","observation_id":"063ad94b-622d-44f6-9392-3df1945fcdd7","resolution":{"observed_at":"2026-05-23T20:03:26.239077Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2405.16281","last_updated":"2024-05-25T15:36:37Z","snapshot_observed_at":"2026-07-06T18:19:56.964336Z","submitted_at":"2024-05-25T15:36:37Z","title":"ConStat: Performance-Based Contamination Detection in Large Language Models","version":1},"cited_work":{"arxiv_id":"2405.16281","doi":"10.48550/arxiv.2405.16281","metadata_source":"arxiv_reference","pith_arxiv_id":"2405.16281","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"ConStat : Performance - Based Contamination Detection in Large Language Models , May 2024","venue":"arXiv (Cornell University)","work_id":"0ac6f3f8-c8f7-475e-b82e-65ab2fcb3a1a","year":2024},"citing_paper":{"arxiv_id":"2410.07095","last_updated":"2025-02-26T11:57:30Z","snapshot_observed_at":"2026-08-06T21:08:58.653035Z","submitted_at":"2024-10-09T17:34:27Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","version":6},"reference_index":7,"source":"arxiv_source","source_observed_at":"2026-05-23T19:11:20.600633Z"},"links":{"cited_paper":"/paper/2405.16281","citing_paper":"/paper/2410.07095"},"observation_digest":"sha256:5c15b264a813d1c194afbdae1e34bd3efcae28325044324d46f8e0fdf1694dc6","observation_id":"bfc1c359-73ad-4401-95e5-ac006cf958c6","resolution":{"observed_at":"2026-05-23T19:13:21.549550Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"GitHub Copilot Workspace : Welcome to the Copilot -native developer environment, April 2024","venue":null,"work_id":"4ddf9426-07b6-4a67-98a8-28b3e9e4111b","year":2024},"citing_paper":{"arxiv_id":"2410.07095","last_updated":"2025-02-26T11:57:30Z","snapshot_observed_at":"2026-08-06T21:08:58.653035Z","submitted_at":"2024-10-09T17:34:27Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","version":6},"reference_index":8,"source":"arxiv_source","source_observed_at":"2026-05-23T19:11:20.600633Z"},"links":{"citing_paper":"/paper/2410.07095"},"observation_digest":"sha256:3676bf8ae5beabc91b1f95e72e08a0d82f911157c20fe4efd3b9ed59b4b44589","observation_id":"0a4c73d4-f7b1-4862-912d-044d64b9f07f","resolution":{"observed_at":"2026-05-23T20:03:26.234961Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Code Droid Technical Report , June 2024","venue":null,"work_id":"f3a8cedb-9fe8-4e2c-bc4b-d31d3540d8ec","year":2024},"citing_paper":{"arxiv_id":"2410.07095","last_updated":"2025-02-26T11:57:30Z","snapshot_observed_at":"2026-08-06T21:08:58.653035Z","submitted_at":"2024-10-09T17:34:27Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","version":6},"reference_index":9,"source":"arxiv_source","source_observed_at":"2026-05-23T19:11:20.600633Z"},"links":{"citing_paper":"/paper/2410.07095"},"observation_digest":"sha256:26dd8b9222baf3118a9a1b9c76e861516e05a946bc3ad4f9d46852d162092e36","observation_id":"91bcb36d-b566-4dc5-85bc-cba24f532b4f","resolution":{"observed_at":"2026-05-23T20:03:26.231007Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.06411","last_updated":"2024-04-09T16:01:24Z","snapshot_observed_at":"2026-07-06T17:57:52.364338Z","submitted_at":"2024-04-09T16:01:24Z","title":"AgentQuest: A Modular Benchmark Framework to Measure Progress and Improve LLM Agents","version":1},"cited_work":{"arxiv_id":"2404.06411","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2404.06411","snapshot_observed_at":"2026-06-29T20:13:58.831517Z","title":"AgentQuest : A Modular Benchmark Framework to Measure Progress and Improve LLM Agents , April 2024","venue":null,"work_id":"a183e18a-f72d-46fd-b6fe-785175cce766","year":2024},"citing_paper":{"arxiv_id":"2410.07095","last_updated":"2025-02-26T11:57:30Z","snapshot_observed_at":"2026-08-06T21:08:58.653035Z","submitted_at":"2024-10-09T17:34:27Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","version":6},"reference_index":10,"source":"arxiv_source","source_observed_at":"2026-05-23T19:11:20.600633Z"},"links":{"cited_paper":"/paper/2404.06411","citing_paper":"/paper/2410.07095"},"observation_digest":"sha256:574dbbb57d97d6ea6ab33fa63a36279d926c8ff66d79fdbe131c1d84b7533fd2","observation_id":"1ad5e1cc-4025-4189-834a-920f3b668c5c","resolution":{"observed_at":"2026-05-23T19:13:21.517261Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Frontier Safety Framework , May 2024","venue":null,"work_id":"f1904b18-462e-434c-ab21-80851a25dfb7","year":2024},"citing_paper":{"arxiv_id":"2410.07095","last_updated":"2025-02-26T11:57:30Z","snapshot_observed_at":"2026-08-06T21:08:58.653035Z","submitted_at":"2024-10-09T17:34:27Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","version":6},"reference_index":11,"source":"arxiv_source","source_observed_at":"2026-05-23T19:11:20.600633Z"},"links":{"citing_paper":"/paper/2410.07095"},"observation_digest":"sha256:02efe8c262513b5afb8f20b5608410f8ff7eeb288f3439e194d8a4094c562506","observation_id":"b17d8e9a-d934-49d7-88d9-6219f8ef0a17","resolution":{"observed_at":"2026-05-23T20:03:26.226806Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2105.09938","last_updated":"2021-11-08T21:16:44Z","snapshot_observed_at":"2026-08-04T23:13:25.514661Z","submitted_at":"2021-05-20T17:58:42Z","title":"Measuring Coding Challenge Competence With APPS","version":3},"cited_work":{"arxiv_id":"2105.09938","doi":"10.18653/v1/2021.eacl-main.225","metadata_source":"pith","pith_arxiv_id":"2105.09938","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Measuring Coding Challenge Competence With APPS","venue":"cs.SE","work_id":"c014c12f-1080-4cb2-ae03-ab6b7c09445c","year":2021},"citing_paper":{"arxiv_id":"2410.07095","last_updated":"2025-02-26T11:57:30Z","snapshot_observed_at":"2026-08-06T21:08:58.653035Z","submitted_at":"2024-10-09T17:34:27Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","version":6},"reference_index":12,"source":"arxiv_source","source_observed_at":"2026-05-23T19:11:20.600633Z"},"links":{"cited_paper":"/paper/2105.09938","citing_paper":"/paper/2410.07095"},"observation_digest":"sha256:3f4e71f8c90e1cca58b753aeca2c15013a9648780a1dbf61d50b7230c115704f","observation_id":"43b31cb2-ca58-4eef-8763-1347390b8110","resolution":{"observed_at":"2026-05-23T19:13:21.521250Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.13010","last_updated":"2024-05-24T11:47:24Z","snapshot_observed_at":"2026-07-06T17:05:56.282078Z","submitted_at":"2023-12-20T13:22:41Z","title":"AgentCoder: Multi-Agent-based Code Generation with Iterative Testing and Optimisation","version":3},"cited_work":{"arxiv_id":"2312.13010","doi":"10.48550/arxiv.2312.13010","metadata_source":"pith","pith_arxiv_id":"2312.13010","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"AgentCoder: Multi-Agent-based Code Generation with Iterative Testing and Optimisation","venue":"cs.CL","work_id":"a59b2f13-e7d4-4d8b-bffd-504f2417da02","year":2023},"citing_paper":{"arxiv_id":"2410.07095","last_updated":"2025-02-26T11:57:30Z","snapshot_observed_at":"2026-08-06T21:08:58.653035Z","submitted_at":"2024-10-09T17:34:27Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","version":6},"reference_index":13,"source":"arxiv_source","source_observed_at":"2026-05-23T19:11:20.600633Z"},"links":{"cited_paper":"/paper/2312.13010","citing_paper":"/paper/2410.07095"},"observation_digest":"sha256:7bccf44968aab06a3e576099248f394ba657c3cc616f297abe816e72cfa4518d","observation_id":"4a595fff-28c7-4351-a9f6-de5ac44b11e9","resolution":{"observed_at":"2026-05-23T19:13:21.526637Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"MLAgentBench : Evaluating Language Agents on Machine Learning Experimentation","venue":null,"work_id":"514600b2-7f6b-4945-a075-7e7b8a697f45","year":2024},"citing_paper":{"arxiv_id":"2410.07095","last_updated":"2025-02-26T11:57:30Z","snapshot_observed_at":"2026-08-06T21:08:58.653035Z","submitted_at":"2024-10-09T17:34:27Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","version":6},"reference_index":14,"source":"arxiv_source","source_observed_at":"2026-05-23T19:11:20.600633Z"},"links":{"citing_paper":"/paper/2410.07095"},"observation_digest":"sha256:a1c33f278f59e573519a14506cbd2067cfb889b499647e00b7081d47f3cdc73b","observation_id":"b0ced4d7-6faf-4949-b66b-16f20112a0c3","resolution":{"observed_at":"2026-05-23T20:03:26.219044Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2403.07974","last_updated":"2024-06-06T17:41:21Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2024-03-12T17:58:04Z","title":"LiveCodeBench: Holistic and Contamination Free Evaluation of Large Language Models for Code","version":2},"cited_work":{"arxiv_id":"2403.07974","doi":"10.1109/icsme52107.2021.00025","metadata_source":"pith","pith_arxiv_id":"2403.07974","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"LiveCodeBench: Holistic and Contamination Free Evaluation of Large Language Models for Code","venue":"cs.SE","work_id":"ea9e51ce-1e75-4182-92d8-4d25f70d2ee4","year":2024},"citing_paper":{"arxiv_id":"2410.07095","last_updated":"2025-02-26T11:57:30Z","snapshot_observed_at":"2026-08-06T21:08:58.653035Z","submitted_at":"2024-10-09T17:34:27Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","version":6},"reference_index":15,"source":"arxiv_source","source_observed_at":"2026-05-23T19:11:20.600633Z"},"links":{"cited_paper":"/paper/2403.07974","citing_paper":"/paper/2410.07095"},"observation_digest":"sha256:9cd86ab2b17b1ccf95e2996303c91609f15ddec365b2f86f105e1059b3aabd69","observation_id":"8db45d11-7b65-4b61-804c-b0b81bf12438","resolution":{"observed_at":"2026-05-23T19:13:21.544518Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.06770","last_updated":"2024-11-11T23:05:04Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-10-10T16:47:29Z","title":"SWE-bench: Can Language Models Resolve Real-World GitHub Issues?","version":3},"cited_work":{"arxiv_id":"2310.06770","doi":"10.1145/512927.512945","metadata_source":"pith","pith_arxiv_id":"2310.06770","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"SWE-bench: Can Language Models Resolve Real-World GitHub Issues?","venue":"cs.CL","work_id":"d0effe15-a689-441a-8e3f-ea35f1c4e4b1","year":2023},"citing_paper":{"arxiv_id":"2410.07095","last_updated":"2025-02-26T11:57:30Z","snapshot_observed_at":"2026-08-06T21:08:58.653035Z","submitted_at":"2024-10-09T17:34:27Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","version":6},"reference_index":16,"source":"arxiv_source","source_observed_at":"2026-05-23T19:11:20.600633Z"},"links":{"cited_paper":"/paper/2310.06770","citing_paper":"/paper/2410.07095"},"observation_digest":"sha256:f657a491361219975574fd657a696375a211593cd4165ad0c46a7cd59b0bafd5","observation_id":"8af70357-a0e3-45bc-929d-a3a2381a43b2","resolution":{"observed_at":"2026-05-23T19:13:21.566443Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.07703","last_updated":"2025-04-11T14:12:58Z","snapshot_observed_at":"2026-08-03T11:14:55.887592Z","submitted_at":"2024-09-12T02:08:00Z","title":"DSBench: How Far Are Data Science Agents from Becoming Data Science Experts?","version":3},"cited_work":{"arxiv_id":"2409.07703","doi":"10.48550/arxiv.2409.07703","metadata_source":"arxiv_reference","pith_arxiv_id":"2409.07703","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"DSBench : How Far Are Data Science Agents to Becoming Data Science Experts ?, September 2024","venue":"arXiv (Cornell University)","work_id":"6b5b5522-b2f4-402e-ac4c-1a209d3d9d60","year":2024},"citing_paper":{"arxiv_id":"2410.07095","last_updated":"2025-02-26T11:57:30Z","snapshot_observed_at":"2026-08-06T21:08:58.653035Z","submitted_at":"2024-10-09T17:34:27Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","version":6},"reference_index":17,"source":"arxiv_source","source_observed_at":"2026-05-23T19:11:20.600633Z"},"links":{"cited_paper":"/paper/2409.07703","citing_paper":"/paper/2410.07095"},"observation_digest":"sha256:c00f79ff399d62f177c0a3f2ec39ff78d0cf6ea28332255edf0ff98aa765b88c","observation_id":"556b8617-b887-420c-9d9f-a9e5dc970094","resolution":{"observed_at":"2026-05-23T19:13:21.510007Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Kaggle Progression System Kaggle","venue":null,"work_id":"c4935f82-ea8f-448e-bb58-0e6f016efdfb","year":2024},"citing_paper":{"arxiv_id":"2410.07095","last_updated":"2025-02-26T11:57:30Z","snapshot_observed_at":"2026-08-06T21:08:58.653035Z","submitted_at":"2024-10-09T17:34:27Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","version":6},"reference_index":18,"source":"arxiv_source","source_observed_at":"2026-05-23T19:11:20.600633Z"},"links":{"citing_paper":"/paper/2410.07095"},"observation_digest":"sha256:566fa74e027c9c6e997def04586e2e0c53d7ca56111da2915c402097320204a7","observation_id":"4fc3763e-a96e-4bdf-9174-369cbc7c776e","resolution":{"observed_at":"2026-05-23T20:03:26.215181Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Research: quantifying GitHub Copilot ’s impact on developer productivity and happiness, September 2022","venue":null,"work_id":"87179a40-8b3b-46c1-9fb4-2d15c87389ea","year":2022},"citing_paper":{"arxiv_id":"2410.07095","last_updated":"2025-02-26T11:57:30Z","snapshot_observed_at":"2026-08-06T21:08:58.653035Z","submitted_at":"2024-10-09T17:34:27Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","version":6},"reference_index":19,"source":"arxiv_source","source_observed_at":"2026-05-23T19:11:20.600633Z"},"links":{"citing_paper":"/paper/2410.07095"},"observation_digest":"sha256:6ad29778bebce101851a199b068b4850919b608cde9106c8374e1f63a5ae2af2","observation_id":"e80d8849-0494-4cf5-97a1-9a33bc0c445a","resolution":{"observed_at":"2026-05-23T20:03:26.211519Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.01502","last_updated":"2024-07-01T17:48:14Z","snapshot_observed_at":"2026-08-06T22:15:08.500389Z","submitted_at":"2024-07-01T17:48:14Z","title":"AI Agents That Matter","version":1},"cited_work":{"arxiv_id":"2407.01502","doi":"10.48550/arxiv.2407.01502","metadata_source":"pith","pith_arxiv_id":"2407.01502","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Siegel, Nitya Nadgir, and Arvind Narayanan","venue":"cs.LG","work_id":"07877e57-5393-47ee-ae5f-563ff8f9a6b2","year":2024},"citing_paper":{"arxiv_id":"2410.07095","last_updated":"2025-02-26T11:57:30Z","snapshot_observed_at":"2026-08-06T21:08:58.653035Z","submitted_at":"2024-10-09T17:34:27Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","version":6},"reference_index":20,"source":"arxiv_source","source_observed_at":"2026-05-23T19:11:20.600633Z"},"links":{"cited_paper":"/paper/2407.01502","citing_paper":"/paper/2410.07095"},"observation_digest":"sha256:f1d37e186147658834ae87f1381f703ce560e89b2b3fce00b6c197783040a793","observation_id":"0d7dd57a-ebc5-419b-a74b-fff200b39d49","resolution":{"observed_at":"2026-05-23T19:13:21.505537Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-04T01:08:07.572928+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-04T01:08:07.572928+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":"10.1126/science.abq1158","metadata_source":"openalex","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mankowitz, Esme Sutherland Robson, Pushmeet Kohli, Nando De Freitas, Koray Kavukcuoglu, and Oriol Vinyals","venue":"Science","work_id":"cc452f34-3d34-41ff-9206-8edad6625ce6","year":2022},"citing_paper":{"arxiv_id":"2410.07095","last_updated":"2025-02-26T11:57:30Z","snapshot_observed_at":"2026-08-06T21:08:58.653035Z","submitted_at":"2024-10-09T17:34:27Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","version":6},"reference_index":21,"source":"arxiv_source","source_observed_at":"2026-05-23T19:11:20.600633Z"},"links":{"citing_paper":"/paper/2410.07095"},"observation_digest":"sha256:8f3b07715a60cf277bf504a9b72f10f7dabbb310cd2a2a93689efb160fd92d55","observation_id":"a88b45dc-14c6-4881-8967-babcd2da0a5d","resolution":{"observed_at":"2026-05-23T19:13:20.478391Z","resolver_source":"doi","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-07-11T15:53:23.512078+00:00","source":"crossref_status_cache"},{"observed_at":"2026-07-11T15:53:23.512078+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2308.03688","last_updated":"2025-10-04T03:54:18Z","snapshot_observed_at":"2026-08-06T20:36:41.418114Z","submitted_at":"2023-08-07T16:08:11Z","title":"AgentBench: Evaluating LLMs as Agents","version":3},"cited_work":{"arxiv_id":"2308.03688","doi":"10.1109/fllm63129.2024.10852426","metadata_source":"pith","pith_arxiv_id":"2308.03688","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"AgentBench: Evaluating LLMs as Agents","venue":"cs.AI","work_id":"a37549b4-4c94-412d-acc4-4efeb08509be","year":2023},"citing_paper":{"arxiv_id":"2410.07095","last_updated":"2025-02-26T11:57:30Z","snapshot_observed_at":"2026-08-06T21:08:58.653035Z","submitted_at":"2024-10-09T17:34:27Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","version":6},"reference_index":22,"source":"arxiv_source","source_observed_at":"2026-05-23T19:11:20.600633Z"},"links":{"cited_paper":"/paper/2308.03688","citing_paper":"/paper/2410.07095"},"observation_digest":"sha256:b51eba4834d43a4ef53b8339a318d4863ae242caaadec2360e344c571ba3de61","observation_id":"129dc745-ebc3-4563-ac56-0c49e86d4d97","resolution":{"observed_at":"2026-05-23T19:13:21.557714Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Vesuvius challenge - ink detection","venue":null,"work_id":"24d3bce2-69c4-4ed5-9b8b-d371a5355821","year":2023},"citing_paper":{"arxiv_id":"2410.07095","last_updated":"2025-02-26T11:57:30Z","snapshot_observed_at":"2026-08-06T21:08:58.653035Z","submitted_at":"2024-10-09T17:34:27Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","version":6},"reference_index":23,"source":"arxiv_source","source_observed_at":"2026-05-23T19:11:20.600633Z"},"links":{"citing_paper":"/paper/2410.07095"},"observation_digest":"sha256:6c0373668621bace89155f501d1945a4a46ebad3d1847ad450bc7d7fe02c0c3e","observation_id":"d67ec729-61b6-428c-999c-2ea8b6b34c40","resolution":{"observed_at":"2026-05-23T20:03:26.207090Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Discovering and exploring cases of educational source code plagiarism with Dolos","venue":null,"work_id":"78a332ee-a71e-4c13-829c-6569a95b9530","year":2024},"citing_paper":{"arxiv_id":"2410.07095","last_updated":"2025-02-26T11:57:30Z","snapshot_observed_at":"2026-08-06T21:08:58.653035Z","submitted_at":"2024-10-09T17:34:27Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","version":6},"reference_index":24,"source":"arxiv_source","source_observed_at":"2026-05-23T19:11:20.600633Z"},"links":{"citing_paper":"/paper/2410.07095"},"observation_digest":"sha256:33edfd91f95146766e0473846f5bfd19d84cc400a1023292d58ab5635cb9dae0","observation_id":"740a3b7a-a86a-4816-8326-2bc938fef9e8","resolution":{"observed_at":"2026-05-23T20:03:26.203336Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.12983","last_updated":"2023-11-21T20:34:47Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2023-11-21T20:34:47Z","title":"GAIA: a benchmark for General AI Assistants","version":1},"cited_work":{"arxiv_id":"2311.12983","doi":"10.48550/arxiv.2311.12983","metadata_source":"pith","pith_arxiv_id":"2311.12983","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"GAIA: a benchmark for General AI Assistants","venue":"cs.CL","work_id":"cf222b33-f7a3-4044-a570-ecfe25edb3f8","year":2023},"citing_paper":{"arxiv_id":"2410.07095","last_updated":"2025-02-26T11:57:30Z","snapshot_observed_at":"2026-08-06T21:08:58.653035Z","submitted_at":"2024-10-09T17:34:27Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","version":6},"reference_index":25,"source":"arxiv_source","source_observed_at":"2026-05-23T19:11:20.600633Z"},"links":{"cited_paper":"/paper/2311.12983","citing_paper":"/paper/2410.07095"},"observation_digest":"sha256:122bae39f9ed063cd469df8d267a36076d77821bb76c6c26181658e3a6cee484","observation_id":"08f8be6f-122d-49e2-832e-4b75c9d471ac","resolution":{"observed_at":"2026-05-23T19:13:21.553404Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-05-20T18:52:17.222295+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-20T18:52:17.222295+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Preparedness Framework , December 2023","venue":null,"work_id":"f62f7151-802e-45b9-a1f0-97cec66f06a7","year":2023},"citing_paper":{"arxiv_id":"2410.07095","last_updated":"2025-02-26T11:57:30Z","snapshot_observed_at":"2026-08-06T21:08:58.653035Z","submitted_at":"2024-10-09T17:34:27Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","version":6},"reference_index":26,"source":"arxiv_source","source_observed_at":"2026-05-23T19:11:20.600633Z"},"links":{"citing_paper":"/paper/2410.07095"},"observation_digest":"sha256:be24de26618d0d5094ccc2ae0d72bc63315fee11a215e043817561d944713ba3","observation_id":"eabc68ac-2f55-445e-9fca-cbee9fb43411","resolution":{"observed_at":"2026-05-23T20:03:26.198351Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Introducing Weco AIDE , April 2024","venue":null,"work_id":"4032f709-24b5-4fb6-b2c5-3f935f5b24eb","year":2024},"citing_paper":{"arxiv_id":"2410.07095","last_updated":"2025-02-26T11:57:30Z","snapshot_observed_at":"2026-08-06T21:08:58.653035Z","submitted_at":"2024-10-09T17:34:27Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","version":6},"reference_index":27,"source":"arxiv_source","source_observed_at":"2026-05-23T19:11:20.600633Z"},"links":{"citing_paper":"/paper/2410.07095"},"observation_digest":"sha256:1650b864bf2c8a77440b7b8b132d7e54471b04bca90f071b5d68249cd06e055d","observation_id":"fb6f2bfa-73e9-4738-933f-8410d816ac27","resolution":{"observed_at":"2026-05-23T20:03:26.194782Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2311.09835","last_updated":"2024-08-21T13:36:30Z","snapshot_observed_at":"2026-07-06T16:48:33.164024Z","submitted_at":"2023-11-16T12:03:21Z","title":"ML-Bench: Evaluating Large Language Models and Agents for Machine Learning Tasks on Repository-Level Code","version":5},"cited_work":{"arxiv_id":"2311.09835","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2311.09835","snapshot_observed_at":"2026-07-01T14:05:46.736275Z","title":"ML - Bench : Evaluating Large Language Models and Agents for Machine Learning Tasks on Repository - Level Code , August 2024","venue":null,"work_id":"c1ece405-4a98-4fe9-90a5-42fa6801339d","year":2023},"citing_paper":{"arxiv_id":"2410.07095","last_updated":"2025-02-26T11:57:30Z","snapshot_observed_at":"2026-08-06T21:08:58.653035Z","submitted_at":"2024-10-09T17:34:27Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","version":6},"reference_index":28,"source":"arxiv_source","source_observed_at":"2026-05-23T19:11:20.600633Z"},"links":{"cited_paper":"/paper/2311.09835","citing_paper":"/paper/2410.07095"},"observation_digest":"sha256:2f24cf6126d76da15166d5df17a18f8054139466ea960dd3357184c050f57f29","observation_id":"230b2e79-42ba-4507-bf2f-bb3b8e302bd2","resolution":{"observed_at":"2026-05-23T19:13:21.535616Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2407.16741","last_updated":"2025-04-18T18:14:31Z","snapshot_observed_at":"2026-08-02T14:58:44.167588Z","submitted_at":"2024-07-23T17:50:43Z","title":"OpenHands: An Open Platform for AI Software Developers as Generalist Agents","version":3},"cited_work":{"arxiv_id":"2407.16741","doi":"10.1145/3718958.3750537","metadata_source":"pith","pith_arxiv_id":"2407.16741","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"OpenHands: An Open Platform for AI Software Developers as Generalist Agents","venue":"cs.SE","work_id":"f1762ea0-e382-4f38-a28c-adc643789859","year":2024},"citing_paper":{"arxiv_id":"2410.07095","last_updated":"2025-02-26T11:57:30Z","snapshot_observed_at":"2026-08-06T21:08:58.653035Z","submitted_at":"2024-10-09T17:34:27Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","version":6},"reference_index":29,"source":"arxiv_source","source_observed_at":"2026-05-23T19:11:20.600633Z"},"links":{"cited_paper":"/paper/2407.16741","citing_paper":"/paper/2410.07095"},"observation_digest":"sha256:500eeb353f7fb6d26c7cb5dbba1e85cd156280d9ac9554c228e0e61b25bf3950","observation_id":"f2d1aebf-c91e-43e6-bca1-a6111eda7e80","resolution":{"observed_at":"2026-05-23T19:13:21.562094Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"The shift from models to compound ai systems","venue":null,"work_id":"354fb11b-bd8f-457e-bdea-144d1aec7d83","year":2024},"citing_paper":{"arxiv_id":"2410.07095","last_updated":"2025-02-26T11:57:30Z","snapshot_observed_at":"2026-08-06T21:08:58.653035Z","submitted_at":"2024-10-09T17:34:27Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","version":6},"reference_index":30,"source":"arxiv_source","source_observed_at":"2026-05-23T19:11:20.600633Z"},"links":{"citing_paper":"/paper/2410.07095"},"observation_digest":"sha256:e4bf366c54642dc8ca037c65047ec6b0dd95baedde78e78f02f7fc94e804d0ed","observation_id":"ef46fc68-42e6-4160-adc9-00f48e3a7f2e","resolution":{"observed_at":"2026-05-23T20:03:26.191266Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2404.05427","last_updated":"2024-07-25T16:54:41Z","snapshot_observed_at":"2026-07-06T17:57:06.330791Z","submitted_at":"2024-04-08T11:55:09Z","title":"AutoCodeRover: Autonomous Program Improvement","version":3},"cited_work":{"arxiv_id":"2404.05427","doi":"10.48550/arxiv.2404.05427","metadata_source":"arxiv_reference","pith_arxiv_id":"2404.05427","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"AutoCodeRover : Autonomous Program Improvement , July 2024","venue":"arXiv (Cornell University)","work_id":"fa93e782-fee4-4c22-9486-d9d5d62f7b93","year":2024},"citing_paper":{"arxiv_id":"2410.07095","last_updated":"2025-02-26T11:57:30Z","snapshot_observed_at":"2026-08-06T21:08:58.653035Z","submitted_at":"2024-10-09T17:34:27Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","version":6},"reference_index":31,"source":"arxiv_source","source_observed_at":"2026-05-23T19:11:20.600633Z"},"links":{"cited_paper":"/paper/2404.05427","citing_paper":"/paper/2410.07095"},"observation_digest":"sha256:a4090204b5613e5f3ada6afce5429a5db43a1dd82c016fe0061672300cf40771","observation_id":"52ade589-ecb0-4e0a-8f4b-9fb84951d706","resolution":{"observed_at":"2026-05-23T19:13:21.530857Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.10970","last_updated":"2023-08-02T03:59:34Z","snapshot_observed_at":"2026-08-05T08:12:59.240898Z","submitted_at":"2023-04-21T14:06:44Z","title":"Can GPT-4 Perform Neural Architecture Search?","version":4},"cited_work":{"arxiv_id":"2304.10970","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2304.10970","snapshot_observed_at":"2026-07-03T04:37:37.181995Z","title":"Can GPT -4 Perform Neural Architecture Search ?, August 2023","venue":null,"work_id":"1ede0a1a-1591-47fe-b0cd-c20c6d02ec03","year":2023},"citing_paper":{"arxiv_id":"2410.07095","last_updated":"2025-02-26T11:57:30Z","snapshot_observed_at":"2026-08-06T21:08:58.653035Z","submitted_at":"2024-10-09T17:34:27Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","version":6},"reference_index":32,"source":"arxiv_source","source_observed_at":"2026-05-23T19:11:20.600633Z"},"links":{"cited_paper":"/paper/2304.10970","citing_paper":"/paper/2410.07095"},"observation_digest":"sha256:088b6b4e41df6a6cf6522663e73eb81df6414cb2577243a0e67a680c9c897d40","observation_id":"a36e9827-74bb-41e2-9818-050d571d840b","resolution":{"observed_at":"2026-05-23T19:13:21.540246Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-09T00:05:48.311803Z","title":"write newline","venue":null,"work_id":"8e5fda61-e601-4df4-8204-015bee341570","year":null},"citing_paper":{"arxiv_id":"2410.07095","last_updated":"2025-02-26T11:57:30Z","snapshot_observed_at":"2026-08-06T21:08:58.653035Z","submitted_at":"2024-10-09T17:34:27Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","version":6},"reference_index":33,"source":"arxiv_source","source_observed_at":"2026-05-23T19:11:20.600633Z"},"links":{"citing_paper":"/paper/2410.07095"},"observation_digest":"sha256:478fe3709b38e8f8aecfec607f276d8ab5889b5c3628538e0e5a8ddd49c4ef1f","observation_id":"6f895805-6347-4eac-800d-13ca628d0fbd","resolution":{"observed_at":"2026-05-23T20:03:26.187338Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-08T23:55:44.012295Z","title":"@esa (Ref","venue":null,"work_id":"b058608d-98d0-4821-a4ae-403d2b7cd411","year":null},"citing_paper":{"arxiv_id":"2410.07095","last_updated":"2025-02-26T11:57:30Z","snapshot_observed_at":"2026-08-06T21:08:58.653035Z","submitted_at":"2024-10-09T17:34:27Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","version":6},"reference_index":34,"source":"arxiv_source","source_observed_at":"2026-05-23T19:11:20.600633Z"},"links":{"citing_paper":"/paper/2410.07095"},"observation_digest":"sha256:f153186cbfb2b948d28fa74f256532fd740fdb5925976e45d21d22ca7d4fc8ba","observation_id":"f847085e-e9ab-47cc-9d4b-77a13c06e902","resolution":{"observed_at":"2026-05-23T20:03:26.243426Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-09T00:05:48.337191Z","title":null,"venue":null,"work_id":"ea79bfb8-d434-45e9-8607-416d3839ec5c","year":null},"citing_paper":{"arxiv_id":"2410.07095","last_updated":"2025-02-26T11:57:30Z","snapshot_observed_at":"2026-08-06T21:08:58.653035Z","submitted_at":"2024-10-09T17:34:27Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","version":6},"reference_index":35,"source":"arxiv_source","source_observed_at":"2026-05-23T19:11:20.600633Z"},"links":{"citing_paper":"/paper/2410.07095"},"observation_digest":"sha256:72e5c5dba3a2f9513aa5bd50ecfca4a5d47e2da66c3cc117c6c524f1e2ada2fa","observation_id":"3db40eef-97a1-40e9-bc8f-41c2a5fa4593","resolution":{"observed_at":"2026-05-23T20:03:26.183215Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":null,"venue":null,"work_id":"d051d6aa-efca-49eb-a66d-8ea01bf21294","year":null},"citing_paper":{"arxiv_id":"2410.07095","last_updated":"2025-02-26T11:57:30Z","snapshot_observed_at":"2026-08-06T21:08:58.653035Z","submitted_at":"2024-10-09T17:34:27Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","version":6},"reference_index":36,"source":"arxiv_source","source_observed_at":"2026-05-23T19:11:20.600633Z"},"links":{"citing_paper":"/paper/2410.07095"},"observation_digest":"sha256:e3f9bbd120e986210c55ae02b3670c825e50f4c9c819f9e3e8bc6bfef30abec9","observation_id":"ccc572ff-0c54-41e8-8e34-f6e8b3856200","resolution":{"observed_at":"2026-05-23T20:03:26.178485Z","resolver_source":"raw_fallback","status":"unresolved"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-07T06:34:17.273281+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2410.07095","last_updated":"2025-02-26T11:57:30Z","latest_version":6,"primary_category":"cs.CL","snapshot_observed_at":"2026-08-06T21:08:58.653035Z","submitted_at":"2024-10-09T17:34:27Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering"},"reference_resolution":{"displayed":36,"state_counts":{"malformed_identifier":0,"metadata_mismatch":3,"parse_uncertain":0,"unresolved":2,"verified_exact":15,"verified_fuzzy":16},"total_outbound_references":36},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-07T06:34:17.273281+00:00","source":"crossref"},{"observed_at":"2026-08-07T06:34:11.927384+00:00","source":"retraction_watch"}],"thesis":"As of 7 August 2026, this Paper Citation Record lists 36 of 36 outbound references and 100 inbound Pith citation observations for arXiv:2410.07095."}