{"as_of":"2026-08-17T18:07:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:f06ace1f983bde793381b1e596e6958c35939b8ccf1e5f1ee0167530263b2656","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":76,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":76,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-17T06:30:58.91139+00:00","state":"measured"},{"denominator":76,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":76,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-16T10:59:33.392950Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"pith","source_observed_at":"2026-07-05T17:51:14.743354Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2310.03302","last_updated":"2024-04-14T21:02:16Z","snapshot_observed_at":"2026-08-17T14:56:37.479640Z","submitted_at":"2023-10-05T04:06:12Z","title":"MLAgentBench: Evaluating Language Agents on Machine Learning Experimentation","version":2},"cited_work":{"arxiv_id":"2310.03302","doi":null,"metadata_source":"pith","pith_arxiv_id":"2310.03302","snapshot_observed_at":"2026-07-05T17:51:14.743354Z","title":"Mlagentbench: Evaluating language agents on ma- chine learning experimentation","venue":"cs.LG","work_id":"5655b20a-fbb7-4a39-8605-d6e1d689895a","year":2023},"citing_paper":{"arxiv_id":"2403.07974","last_updated":"2024-06-06T17:41:21Z","snapshot_observed_at":"2026-08-16T07:05:57.323612Z","submitted_at":"2024-03-12T17:58:04Z","title":"LiveCodeBench: Holistic and Contamination Free Evaluation of Large Language Models for Code","version":2},"reference_index":164,"source":"arxiv_source","source_observed_at":"2026-05-10T17:34:42.565806Z"},"links":{"cited_paper":"/paper/2310.03302","citing_paper":"/paper/2403.07974"},"observation_digest":"sha256:f8fb0e5351ef9c01e0fbd012347e841c7b92c76dbaa41c3793e3b10c05edd2a3","observation_id":"e22682f9-6d22-4950-8ab2-4df6f2cbf0ab","resolution":{"observed_at":"2026-05-10T17:34:42.767663Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.03302","last_updated":"2024-04-14T21:02:16Z","snapshot_observed_at":"2026-08-17T14:56:37.479640Z","submitted_at":"2023-10-05T04:06:12Z","title":"MLAgentBench: Evaluating Language Agents on Machine Learning Experimentation","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.03302","snapshot_observed_at":"2026-08-12T17:12:27.467241Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2411.12828","last_updated":"2024-11-19T19:33:16Z","snapshot_observed_at":"2026-08-12T17:07:15.117769Z","submitted_at":"2024-11-19T19:33:16Z","title":"Probing the Capacity of Language Model Agents to Operationalize Disparate Experiential Context Despite Distraction","version":1},"reference_index":5,"source":"arxiv_source","source_observed_at":"2026-08-12T17:12:27.467241Z"},"links":{"cited_paper":"/paper/2310.03302","citing_paper":"/paper/2411.12828"},"observation_digest":"sha256:85b7bf3a8e1f75792b209423c05e4ede9303730b9a66d83080c33998f5bb384c","observation_id":"dd32d7fe-6786-494d-9a52-773dd83e9a5c","resolution":{"observed_at":"2026-08-12T17:12:27.467241Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.03302","last_updated":"2024-04-14T21:02:16Z","snapshot_observed_at":"2026-08-17T14:56:37.479640Z","submitted_at":"2023-10-05T04:06:12Z","title":"MLAgentBench: Evaluating Language Agents on Machine Learning Experimentation","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.03302","snapshot_observed_at":"2026-08-12T14:33:13.146051Z","title":"& Leskovec, J","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2411.15114","last_updated":"2025-05-27T03:32:23Z","snapshot_observed_at":"2026-08-12T16:49:37.896696Z","submitted_at":"2024-11-22T18:30:46Z","title":"RE-Bench: Evaluating frontier AI R&D capabilities of language model agents against human experts","version":2},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-08-12T14:33:13.146051Z"},"links":{"cited_paper":"/paper/2310.03302","citing_paper":"/paper/2411.15114"},"observation_digest":"sha256:52b67f054e12aeaf4811aa960da3ece681b90c1446f5d827f605ebc4843a5c02","observation_id":"482e5e49-c928-48db-80dd-957436abcce4","resolution":{"observed_at":"2026-08-12T14:33:13.146051Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.03302","last_updated":"2024-04-14T21:02:16Z","snapshot_observed_at":"2026-08-17T14:56:37.479640Z","submitted_at":"2023-10-05T04:06:12Z","title":"MLAgentBench: Evaluating Language Agents on Machine Learning Experimentation","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.03302","snapshot_observed_at":"2026-08-11T17:03:41.279896Z","title":"Mlagentbench: Evaluating language agents on machine learning experimentation.arXiv preprint arXiv:2310.03302, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2412.09529","last_updated":"2025-04-08T01:25:47Z","snapshot_observed_at":"2026-08-17T05:27:35.820649Z","submitted_at":"2024-12-12T18:20:16Z","title":"How Well Can Modern LLMs Act as Agent Cores in Radiology Environments?","version":3},"reference_index":51,"source":"pdf_text","source_observed_at":"2026-08-11T17:03:41.279896Z"},"links":{"cited_paper":"/paper/2310.03302","citing_paper":"/paper/2412.09529"},"observation_digest":"sha256:8ad5d54254632df8253ecfb2af4104fc72015731eb21f2e6aa35af079e9c8e66","observation_id":"2e654b03-1055-4c8e-81a8-94424a24131e","resolution":{"observed_at":"2026-08-11T17:03:41.279896Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.03302","last_updated":"2024-04-14T21:02:16Z","snapshot_observed_at":"2026-08-17T14:56:37.479640Z","submitted_at":"2023-10-05T04:06:12Z","title":"MLAgentBench: Evaluating Language Agents on Machine Learning Experimentation","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.03302","snapshot_observed_at":"2026-08-11T12:33:41.068131Z","title":"Benchmarking large lan- guage models as ai research agents,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2412.14056","last_updated":"2024-12-18T17:06:21Z","snapshot_observed_at":"2026-08-13T23:42:32.185607Z","submitted_at":"2024-12-18T17:06:21Z","title":"A Review of Multimodal Explainable Artificial Intelligence: Past, Present and Future","version":1},"reference_index":243,"source":"pdf_text","source_observed_at":"2026-08-11T12:33:41.068131Z"},"links":{"cited_paper":"/paper/2310.03302","citing_paper":"/paper/2412.14056"},"observation_digest":"sha256:528e27a1f01704dc078275655ab57d252b5854b3ee3599f6ff9d47aad3fb0c2e","observation_id":"de8015ab-8bd0-4a98-80ea-3f67f0c5bd17","resolution":{"observed_at":"2026-08-11T12:33:41.068131Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.03302","last_updated":"2024-04-14T21:02:16Z","snapshot_observed_at":"2026-08-17T14:56:37.479640Z","submitted_at":"2023-10-05T04:06:12Z","title":"MLAgentBench: Evaluating Language Agents on Machine Learning Experimentation","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.03302","snapshot_observed_at":"2026-08-11T05:29:24.434408Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2412.17481","last_updated":"2025-01-07T12:48:22Z","snapshot_observed_at":"2026-08-15T07:26:49.037180Z","submitted_at":"2024-12-23T11:11:51Z","title":"A Survey on LLM-based Multi-Agent System: Recent Advances and New Frontiers in Application","version":2},"reference_index":31,"source":"arxiv_source","source_observed_at":"2026-08-11T05:29:24.434408Z"},"links":{"cited_paper":"/paper/2310.03302","citing_paper":"/paper/2412.17481"},"observation_digest":"sha256:0db41417ff04d711f10c07a662decdf7b1959d0445e94c6c8f64dcf6dbcd91f6","observation_id":"2362b601-a1e3-4c51-bafb-9189be1bc3b5","resolution":{"observed_at":"2026-08-11T05:29:24.434408Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.03302","last_updated":"2024-04-14T21:02:16Z","snapshot_observed_at":"2026-08-17T14:56:37.479640Z","submitted_at":"2023-10-05T04:06:12Z","title":"MLAgentBench: Evaluating Language Agents on Machine Learning Experimentation","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.03302","snapshot_observed_at":"2026-08-09T19:46:23.635504Z","title":"Mlagentbench: Eval- uating language agents on machine learning experimentation","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2502.00226","last_updated":"2025-01-31T23:47:02Z","snapshot_observed_at":"2026-08-16T14:03:27.509519Z","submitted_at":"2025-01-31T23:47:02Z","title":"HackerRank-ASTRA: Evaluating Correctness & Consistency of Large Language Models on cross-domain multi-file project problems","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-09T19:46:23.635504Z"},"links":{"cited_paper":"/paper/2310.03302","citing_paper":"/paper/2502.00226"},"observation_digest":"sha256:5ada145ea84a9c4476b50c9014752667d70996cef7eb74b2332796b5736daf19","observation_id":"0b1e7f17-7f48-4c2c-952d-dd87aa842518","resolution":{"observed_at":"2026-08-09T19:46:23.635504Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.03302","last_updated":"2024-04-14T21:02:16Z","snapshot_observed_at":"2026-08-17T14:56:37.479640Z","submitted_at":"2023-10-05T04:06:12Z","title":"MLAgentBench: Evaluating Language Agents on Machine Learning Experimentation","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.03302","snapshot_observed_at":"2026-08-16T10:59:33.392950Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2504.16728","last_updated":"2025-05-24T12:40:05Z","snapshot_observed_at":"2026-08-17T14:57:24.659289Z","submitted_at":"2025-04-23T14:01:36Z","title":"IRIS: Interactive Research Ideation System for Accelerating Scientific Discovery","version":2},"reference_index":21,"source":"arxiv_source","source_observed_at":"2026-08-16T10:59:33.392950Z"},"links":{"cited_paper":"/paper/2310.03302","citing_paper":"/paper/2504.16728"},"observation_digest":"sha256:3c83df011301bb166498249a647f1c29278942abb415dc55f14a9f9c8a8fc1af","observation_id":"06221fec-513e-4339-bd7c-6c3668e36722","resolution":{"observed_at":"2026-08-16T10:59:33.392950Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.03302","last_updated":"2024-04-14T21:02:16Z","snapshot_observed_at":"2026-08-17T14:56:37.479640Z","submitted_at":"2023-10-05T04:06:12Z","title":"MLAgentBench: Evaluating Language Agents on Machine Learning Experimentation","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.03302","snapshot_observed_at":"2026-08-15T20:12:30.978625Z","title":"Mlagentbench: Evaluating language agents on machine learning experimentation","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.13941","last_updated":"2025-05-20T05:20:53Z","snapshot_observed_at":"2026-08-17T14:57:33.654987Z","submitted_at":"2025-05-20T05:20:53Z","title":"MLZero: A Multi-Agent System for End-to-end Machine Learning Automation","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-08-15T20:12:30.978625Z"},"links":{"cited_paper":"/paper/2310.03302","citing_paper":"/paper/2505.13941"},"observation_digest":"sha256:da7dd5dab9203341030d3cbe83fde595b723e0f25ae6dde7f8b5da6c29ff9906","observation_id":"33a3b63b-61a8-4645-be24-bb57c4251e68","resolution":{"observed_at":"2026-08-15T20:12:30.978625Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.03302","last_updated":"2024-04-14T21:02:16Z","snapshot_observed_at":"2026-08-17T14:56:37.479640Z","submitted_at":"2023-10-05T04:06:12Z","title":"MLAgentBench: Evaluating Language Agents on Machine Learning Experimentation","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.03302","snapshot_observed_at":"2026-08-07T15:42:43.182959Z","title":"Benchmarking large language models as AI research agents.CoRR, abs/2310.03302, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.14148","last_updated":"2025-05-20T09:55:31Z","snapshot_observed_at":"2026-08-16T01:06:47.864078Z","submitted_at":"2025-05-20T09:55:31Z","title":"MM-Agent: LLM as Agents for Real-world Mathematical Modeling Problem","version":1},"reference_index":50,"source":"pdf_text","source_observed_at":"2026-08-07T15:42:43.182959Z"},"links":{"cited_paper":"/paper/2310.03302","citing_paper":"/paper/2505.14148"},"observation_digest":"sha256:5c82e51cc38562dc438e49a6cf9952c03ca2c3fdd81af37f188deeccac1ec4f9","observation_id":"a9e4d109-9fac-4863-ae07-f3aa906b5f2d","resolution":{"observed_at":"2026-08-07T15:42:43.182959Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.03302","last_updated":"2024-04-14T21:02:16Z","snapshot_observed_at":"2026-08-17T14:56:37.479640Z","submitted_at":"2023-10-05T04:06:12Z","title":"MLAgentBench: Evaluating Language Agents on Machine Learning Experimentation","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.03302","snapshot_observed_at":"2026-08-07T14:46:02.358310Z","title":"arXiv preprint arXiv:2310.03302","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2505.17705","last_updated":"2025-05-23T10:16:16Z","snapshot_observed_at":"2026-08-17T14:57:16.738433Z","submitted_at":"2025-05-23T10:16:16Z","title":"CIKT: A Collaborative and Iterative Knowledge Tracing Framework with Large Language Models","version":1},"reference_index":2023,"source":"pdf_text","source_observed_at":"2026-08-07T14:46:02.358310Z"},"links":{"cited_paper":"/paper/2310.03302","citing_paper":"/paper/2505.17705"},"observation_digest":"sha256:028c8d98823672d5109e7ffd67430efb144b09dc844a88e308995a93ed38bf9c","observation_id":"497203fd-20c9-4fac-8251-62b635e883c4","resolution":{"observed_at":"2026-08-07T14:46:02.358310Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.03302","last_updated":"2024-04-14T21:02:16Z","snapshot_observed_at":"2026-08-17T14:56:37.479640Z","submitted_at":"2023-10-05T04:06:12Z","title":"MLAgentBench: Evaluating Language Agents on Machine Learning Experimentation","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.03302","snapshot_observed_at":"2026-08-07T13:53:16.028495Z","title":"Mlagentbench: Evaluating language agents on machine learning experimentation","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.21577","last_updated":"2025-08-25T13:40:36Z","snapshot_observed_at":"2026-08-17T02:26:20.552658Z","submitted_at":"2025-05-27T08:35:05Z","title":"RepoMaster: Autonomous Exploration and Understanding of GitHub Repositories for Complex Task Solving","version":3},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-08-07T13:53:16.028495Z"},"links":{"cited_paper":"/paper/2310.03302","citing_paper":"/paper/2505.21577"},"observation_digest":"sha256:ea57eff61f9a2b4b4cb09adb4123b53ff08b3b3d1dfe850225111c55f27d0581","observation_id":"36be6c58-142b-41d6-af9d-7998f68f4ee6","resolution":{"observed_at":"2026-08-07T13:53:16.028495Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.03302","last_updated":"2024-04-14T21:02:16Z","snapshot_observed_at":"2026-08-17T14:56:37.479640Z","submitted_at":"2023-10-05T04:06:12Z","title":"MLAgentBench: Evaluating Language Agents on Machine Learning Experimentation","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.03302","snapshot_observed_at":"2026-08-07T13:27:34.074882Z","title":"Benchmarking Large Language Models As AI Research Agents","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2505.21898","last_updated":"2025-05-28T02:23:53Z","snapshot_observed_at":"2026-08-15T13:47:08.166428Z","submitted_at":"2025-05-28T02:23:53Z","title":"Co-Saving: Resource Aware Multi-Agent Collaboration for Software Development","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-07T13:27:34.074882Z"},"links":{"cited_paper":"/paper/2310.03302","citing_paper":"/paper/2505.21898"},"observation_digest":"sha256:4b91e7d3ee88ffd0d59972627b8c906c1ac38482352916c4fae17a60a8ba52c3","observation_id":"5cc0eecf-07f5-42c9-8155-d58f6c755ec1","resolution":{"observed_at":"2026-08-07T13:27:34.074882Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.03302","last_updated":"2024-04-14T21:02:16Z","snapshot_observed_at":"2026-08-17T14:56:37.479640Z","submitted_at":"2023-10-05T04:06:12Z","title":"MLAgentBench: Evaluating Language Agents on Machine Learning Experimentation","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.03302","snapshot_observed_at":"2026-08-07T12:05:55.906410Z","title":"Mlagentbench: Evaluating language agents on machine learning experimentation","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.00794","last_updated":"2025-06-01T02:46:31Z","snapshot_observed_at":"2026-08-12T14:14:03.626008Z","submitted_at":"2025-06-01T02:46:31Z","title":"Predicting Empirical AI Research Outcomes with Language Models","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-08-07T12:05:55.906410Z"},"links":{"cited_paper":"/paper/2310.03302","citing_paper":"/paper/2506.00794"},"observation_digest":"sha256:1e427204cadc76f3fb3e5c080707e6fede70d816259e7986679c50397842ed8e","observation_id":"cdc97199-e562-4ef9-abd7-04dff8b60230","resolution":{"observed_at":"2026-08-07T12:05:55.906410Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.03302","last_updated":"2024-04-14T21:02:16Z","snapshot_observed_at":"2026-08-17T14:56:37.479640Z","submitted_at":"2023-10-05T04:06:12Z","title":"MLAgentBench: Evaluating Language Agents on Machine Learning Experimentation","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.03302","snapshot_observed_at":"2026-08-07T10:22:59.699319Z","title":"Mlagentbench: Evaluating language agents on machine learning experimentation, 2024 a","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.05542","last_updated":"2025-06-05T19:44:38Z","snapshot_observed_at":"2026-08-13T03:20:34.930786Z","submitted_at":"2025-06-05T19:44:38Z","title":"Agentomics-ML: Autonomous Machine Learning Experimentation Agent for Genomic and Transcriptomic Data","version":1},"reference_index":9,"source":"arxiv_source","source_observed_at":"2026-08-07T10:22:59.699319Z"},"links":{"cited_paper":"/paper/2310.03302","citing_paper":"/paper/2506.05542"},"observation_digest":"sha256:5a60937056109b52f9531fa1b8e9052b68f9e7a7c4e12587ce2d22dac9478dfe","observation_id":"c0913741-ea1a-45d3-8443-1b3869f10af8","resolution":{"observed_at":"2026-08-07T10:22:59.699319Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.03302","last_updated":"2024-04-14T21:02:16Z","snapshot_observed_at":"2026-08-17T14:56:37.479640Z","submitted_at":"2023-10-05T04:06:12Z","title":"MLAgentBench: Evaluating Language Agents on Machine Learning Experimentation","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.03302","snapshot_observed_at":"2026-08-07T00:55:16.903112Z","title":"Huang, J","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.12312","last_updated":"2025-06-14T02:22:28Z","snapshot_observed_at":"2026-08-13T14:33:19.317608Z","submitted_at":"2025-06-14T02:22:28Z","title":"Perspective on Utilizing Foundation Models for Laboratory Automation in Materials Research","version":1},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-08-07T00:55:16.903112Z"},"links":{"cited_paper":"/paper/2310.03302","citing_paper":"/paper/2506.12312"},"observation_digest":"sha256:80a4eba6344980e4b48e27c0736df1d1b51ae56397e817f092a9a9441132b4d6","observation_id":"9240ddd9-22e9-456f-af9c-378e3fe489c8","resolution":{"observed_at":"2026-08-07T00:55:16.903112Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.03302","last_updated":"2024-04-14T21:02:16Z","snapshot_observed_at":"2026-08-17T14:56:37.479640Z","submitted_at":"2023-10-05T04:06:12Z","title":"MLAgentBench: Evaluating Language Agents on Machine Learning Experimentation","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.03302","snapshot_observed_at":"2026-08-15T19:28:26.876986Z","title":"Mlagentbench: Evaluating language agents on machine learning experimentation","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.16499","last_updated":"2025-06-19T17:53:28Z","snapshot_observed_at":"2026-08-15T19:22:44.446061Z","submitted_at":"2025-06-19T17:53:28Z","title":"ML-Master: Towards AI-for-AI via Integration of Exploration and Reasoning","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-08-15T19:28:26.876986Z"},"links":{"cited_paper":"/paper/2310.03302","citing_paper":"/paper/2506.16499"},"observation_digest":"sha256:b05dae91f9ddb36bafe5b23de92091391cb06378cf60e0b61401fd55497866e8","observation_id":"824bc8cf-a92d-44e2-b1bf-97ae87bf6406","resolution":{"observed_at":"2026-08-15T19:28:26.876986Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.03302","last_updated":"2024-04-14T21:02:16Z","snapshot_observed_at":"2026-08-17T14:56:37.479640Z","submitted_at":"2023-10-05T04:06:12Z","title":"MLAgentBench: Evaluating Language Agents on Machine Learning Experimentation","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.03302","snapshot_observed_at":"2026-08-06T23:26:56.670441Z","title":"Mlagentbench: Evaluating language agents on machine learning experimentation","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2506.18096","last_updated":"2025-09-03T15:32:23Z","snapshot_observed_at":"2026-08-15T20:12:18.032530Z","submitted_at":"2025-06-22T16:52:48Z","title":"Deep Research Agents: A Systematic Examination And Roadmap","version":2},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-08-06T23:26:56.670441Z"},"links":{"cited_paper":"/paper/2310.03302","citing_paper":"/paper/2506.18096"},"observation_digest":"sha256:ca5058f4329b899c5f3064f1d2d5518489c55b7ac348a0dfb27c2e53e18a473d","observation_id":"34a0a56a-b63d-4394-94e3-d06fd10fc306","resolution":{"observed_at":"2026-08-06T23:26:56.670441Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.03302","last_updated":"2024-04-14T21:02:16Z","snapshot_observed_at":"2026-08-17T14:56:37.479640Z","submitted_at":"2023-10-05T04:06:12Z","title":"MLAgentBench: Evaluating Language Agents on Machine Learning Experimentation","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.03302","snapshot_observed_at":"2026-08-06T22:24:47.622941Z","title":"Mlagentbench: Evaluating language agents on machine learning experimentation","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2506.21763","last_updated":"2025-07-21T06:49:51Z","snapshot_observed_at":"2026-08-14T13:19:32.407341Z","submitted_at":"2025-06-26T20:44:51Z","title":"THE-Tree: Can Tracing Historical Evolution Enhance Scientific Verification and Reasoning?","version":2},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-08-06T22:24:47.622941Z"},"links":{"cited_paper":"/paper/2310.03302","citing_paper":"/paper/2506.21763"},"observation_digest":"sha256:bc4de2370a989d58742e601db98044ed4486401ec5cb838e3138db298a949b74","observation_id":"4af3d1df-4469-45d9-87ff-caa730185580","resolution":{"observed_at":"2026-08-06T22:24:47.622941Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.03302","last_updated":"2024-04-14T21:02:16Z","snapshot_observed_at":"2026-08-17T14:56:37.479640Z","submitted_at":"2023-10-05T04:06:12Z","title":"MLAgentBench: Evaluating Language Agents on Machine Learning Experimentation","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.03302","snapshot_observed_at":"2026-08-06T14:57:06.640490Z","title":"Mlagentbench: Evaluating language agents on machine learning experimentation.arXiv preprint arXiv:2310.03302, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2507.17257","last_updated":"2025-07-23T06:56:15Z","snapshot_observed_at":"2026-08-06T18:04:24.113657Z","submitted_at":"2025-07-23T06:56:15Z","title":"Agent Identity Evals: Measuring Agentic Identity","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-08-06T14:57:06.640490Z"},"links":{"cited_paper":"/paper/2310.03302","citing_paper":"/paper/2507.17257"},"observation_digest":"sha256:130220db2f3075ffff53fd9f9b31e17000b72e05a6ece17d7d9339d4df32c615","observation_id":"43152c97-83f1-4231-afce-e6da05582938","resolution":{"observed_at":"2026-08-06T14:57:06.640490Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.03302","last_updated":"2024-04-14T21:02:16Z","snapshot_observed_at":"2026-08-17T14:56:37.479640Z","submitted_at":"2023-10-05T04:06:12Z","title":"MLAgentBench: Evaluating Language Agents on Machine Learning Experimentation","version":2},"cited_work":{"arxiv_id":"2310.03302","doi":null,"metadata_source":"pith","pith_arxiv_id":"2310.03302","snapshot_observed_at":"2026-07-05T17:51:14.743354Z","title":"Mlagentbench: Evaluating language agents on ma- chine learning experimentation","venue":"cs.LG","work_id":"5655b20a-fbb7-4a39-8605-d6e1d689895a","year":2023},"citing_paper":{"arxiv_id":"2508.10177","last_updated":"2026-04-22T18:28:23Z","snapshot_observed_at":"2026-08-12T17:56:51.477030Z","submitted_at":"2025-08-13T20:29:56Z","title":"KompeteAI: Accelerated Autonomous Multi-Agent System for End-to-End Pipeline Generation for Machine Learning Problems","version":3},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-05-18T22:22:19.478156Z"},"links":{"cited_paper":"/paper/2310.03302","citing_paper":"/paper/2508.10177"},"observation_digest":"sha256:980ecdbab844738f8e41390070012d119a10379deacafccc2e9c84aa2454f923","observation_id":"a62dac5d-51ee-4299-b37e-9688afd25f0a","resolution":{"observed_at":"2026-05-18T22:22:51.733291Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.03302","last_updated":"2024-04-14T21:02:16Z","snapshot_observed_at":"2026-08-17T14:56:37.479640Z","submitted_at":"2023-10-05T04:06:12Z","title":"MLAgentBench: Evaluating Language Agents on Machine Learning Experimentation","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.03302","snapshot_observed_at":"2026-08-05T12:24:02.231780Z","title":"Mlagentbench: Evaluating language agents on machine learning experimentation","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2509.01684","last_updated":"2025-09-01T18:04:10Z","snapshot_observed_at":"2026-08-16T16:10:35.045548Z","submitted_at":"2025-09-01T18:04:10Z","title":"Reinforcement Learning for Machine Learning Engineering Agents","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-08-05T12:24:02.231780Z"},"links":{"cited_paper":"/paper/2310.03302","citing_paper":"/paper/2509.01684"},"observation_digest":"sha256:279fa45afff261f58b487e95fb6e61524d2694c7ac898dc4e65f864dfa9b120c","observation_id":"47afa8af-5f77-41d0-8d91-daac13edd68a","resolution":{"observed_at":"2026-08-05T12:24:02.231780Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.03302","last_updated":"2024-04-14T21:02:16Z","snapshot_observed_at":"2026-08-17T14:56:37.479640Z","submitted_at":"2023-10-05T04:06:12Z","title":"MLAgentBench: Evaluating Language Agents on Machine Learning Experimentation","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.03302","snapshot_observed_at":"2026-08-04T19:21:20.629076Z","title":"arXiv:2310.03302","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2509.09321","last_updated":"2025-09-11T10:10:48Z","snapshot_observed_at":"2026-08-14T23:09:12.344345Z","submitted_at":"2025-09-11T10:10:48Z","title":"Towards Adaptive ML Benchmarks: Web-Agent-Driven Construction, Domain Expansion, and Metric Optimization","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-08-04T19:21:20.629076Z"},"links":{"cited_paper":"/paper/2310.03302","citing_paper":"/paper/2509.09321"},"observation_digest":"sha256:45bf7f708fc9ec5446f5de2b9c62fc4cc415a4d6c43c72c71e5a06d4ecf81eff","observation_id":"227d29d8-f707-437a-8499-71de11a6b9e8","resolution":{"observed_at":"2026-08-04T19:21:20.629076Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.03302","last_updated":"2024-04-14T21:02:16Z","snapshot_observed_at":"2026-08-17T14:56:37.479640Z","submitted_at":"2023-10-05T04:06:12Z","title":"MLAgentBench: Evaluating Language Agents on Machine Learning Experimentation","version":2},"cited_work":{"arxiv_id":"2310.03302","doi":null,"metadata_source":"pith","pith_arxiv_id":"2310.03302","snapshot_observed_at":"2026-07-05T17:51:14.743354Z","title":"Mlagentbench: Evaluating language agents on ma- chine learning experimentation","venue":"cs.LG","work_id":"5655b20a-fbb7-4a39-8605-d6e1d689895a","year":2023},"citing_paper":{"arxiv_id":"2601.05930","last_updated":"2026-04-07T04:58:28Z","snapshot_observed_at":"2026-08-16T17:39:11.866981Z","submitted_at":"2026-01-09T16:44:17Z","title":"Can We Predict Before Executing Machine Learning Agents?","version":2},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-16T15:42:25.852911Z"},"links":{"cited_paper":"/paper/2310.03302","citing_paper":"/paper/2601.05930"},"observation_digest":"sha256:aec553cbb6e8c71822a582a85edf336d0b66d838aff2d96c5208a8deb89c1fc9","observation_id":"19e65b4c-d119-4f78-a9e3-fd558d2d7f9d","resolution":{"observed_at":"2026-05-16T15:43:03.592807Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.03302","last_updated":"2024-04-14T21:02:16Z","snapshot_observed_at":"2026-08-17T14:56:37.479640Z","submitted_at":"2023-10-05T04:06:12Z","title":"MLAgentBench: Evaluating Language Agents on Machine Learning Experimentation","version":2},"cited_work":{"arxiv_id":"2310.03302","doi":null,"metadata_source":"pith","pith_arxiv_id":"2310.03302","snapshot_observed_at":"2026-07-05T17:51:14.743354Z","title":"Mlagentbench: Evaluating language agents on ma- chine learning experimentation","venue":"cs.LG","work_id":"5655b20a-fbb7-4a39-8605-d6e1d689895a","year":2023},"citing_paper":{"arxiv_id":"2602.07906","last_updated":"2026-05-07T09:44:38Z","snapshot_observed_at":"2026-08-16T03:25:54.487595Z","submitted_at":"2026-02-08T10:55:03Z","title":"AceGRPO: Adaptive Curriculum Enhanced Group Relative Policy Optimization for Autonomous Machine Learning Engineering","version":5},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-05-16T06:32:22.038300Z"},"links":{"cited_paper":"/paper/2310.03302","citing_paper":"/paper/2602.07906"},"observation_digest":"sha256:14e8b64fc833dbdf8d8818ab1c85e637b50c2892afb02b09b8bd25f70a1c1fd9","observation_id":"b4090343-b136-41b7-b2cd-14bd4c35b2d0","resolution":{"observed_at":"2026-05-16T06:32:27.339639Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.03302","last_updated":"2024-04-14T21:02:16Z","snapshot_observed_at":"2026-08-17T14:56:37.479640Z","submitted_at":"2023-10-05T04:06:12Z","title":"MLAgentBench: Evaluating Language Agents on Machine Learning Experimentation","version":2},"cited_work":{"arxiv_id":"2310.03302","doi":null,"metadata_source":"pith","pith_arxiv_id":"2310.03302","snapshot_observed_at":"2026-07-05T17:51:14.743354Z","title":"Mlagentbench: Evaluating language agents on ma- chine learning experimentation","venue":"cs.LG","work_id":"5655b20a-fbb7-4a39-8605-d6e1d689895a","year":2023},"citing_paper":{"arxiv_id":"2603.01692","last_updated":"2026-04-12T09:20:27Z","snapshot_observed_at":"2026-08-16T20:18:38.350822Z","submitted_at":"2026-03-02T10:22:47Z","title":"Reasoning as Gradient: Scaling MLE Agents Beyond Tree Search","version":3},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-05-15T17:49:46.383559Z"},"links":{"cited_paper":"/paper/2310.03302","citing_paper":"/paper/2603.01692"},"observation_digest":"sha256:b4751834c92c65df95ee8c3fc7ea71726a8c61b370de9ca117f5ced9bf1dd499","observation_id":"d2b289d4-6ea3-4b36-b315-08cd9ed0c454","resolution":{"observed_at":"2026-05-15T17:50:12.229176Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.03302","last_updated":"2024-04-14T21:02:16Z","snapshot_observed_at":"2026-08-17T14:56:37.479640Z","submitted_at":"2023-10-05T04:06:12Z","title":"MLAgentBench: Evaluating Language Agents on Machine Learning Experimentation","version":2},"cited_work":{"arxiv_id":"2310.03302","doi":null,"metadata_source":"pith","pith_arxiv_id":"2310.03302","snapshot_observed_at":"2026-07-05T17:51:14.743354Z","title":"Mlagentbench: Evaluating language agents on ma- chine learning experimentation","venue":"cs.LG","work_id":"5655b20a-fbb7-4a39-8605-d6e1d689895a","year":2023},"citing_paper":{"arxiv_id":"2604.09791","last_updated":"2026-04-10T18:13:09Z","snapshot_observed_at":"2026-08-17T10:00:31.904489Z","submitted_at":"2026-04-10T18:13:09Z","title":"Pioneer Agent: Continual Improvement of Small Language Models in Production","version":1},"reference_index":41,"source":"arxiv_source","source_observed_at":"2026-05-10T17:48:40.520740Z"},"links":{"cited_paper":"/paper/2310.03302","citing_paper":"/paper/2604.09791"},"observation_digest":"sha256:c84ada8f114c4763bbf0f573c38f7d72af8d1ef779c3962c9c3d49cbb03834e1","observation_id":"dd51afc6-27d5-4d56-9342-87524914d2e2","resolution":{"observed_at":"2026-05-11T06:05:57.368134Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.03302","last_updated":"2024-04-14T21:02:16Z","snapshot_observed_at":"2026-08-17T14:56:37.479640Z","submitted_at":"2023-10-05T04:06:12Z","title":"MLAgentBench: Evaluating Language Agents on Machine Learning Experimentation","version":2},"cited_work":{"arxiv_id":"2310.03302","doi":null,"metadata_source":"pith","pith_arxiv_id":"2310.03302","snapshot_observed_at":"2026-07-05T17:51:14.743354Z","title":"Mlagentbench: Evaluating language agents on ma- chine learning experimentation","venue":"cs.LG","work_id":"5655b20a-fbb7-4a39-8605-d6e1d689895a","year":2023},"citing_paper":{"arxiv_id":"2604.10718","last_updated":"2026-04-12T16:28:51Z","snapshot_observed_at":"2026-08-15T11:49:03.806779Z","submitted_at":"2026-04-12T16:28:51Z","title":"SciPredict: Can LLMs Predict the Outcomes of Scientific Experiments in Natural Sciences?","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-05-10T15:55:34.768853Z"},"links":{"cited_paper":"/paper/2310.03302","citing_paper":"/paper/2604.10718"},"observation_digest":"sha256:8a0b975fb3f6a8de43ed8196fb4ec7e9cc50b9dd6de626adee46ad40fe1f3081","observation_id":"cd3f2020-b5ed-4450-bbe0-7742b0f2ee6c","resolution":{"observed_at":"2026-05-11T09:36:03.453162Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.03302","last_updated":"2024-04-14T21:02:16Z","snapshot_observed_at":"2026-08-17T14:56:37.479640Z","submitted_at":"2023-10-05T04:06:12Z","title":"MLAgentBench: Evaluating Language Agents on Machine Learning Experimentation","version":2},"cited_work":{"arxiv_id":"2310.03302","doi":null,"metadata_source":"pith","pith_arxiv_id":"2310.03302","snapshot_observed_at":"2026-07-05T17:51:14.743354Z","title":"Mlagentbench: Evaluating language agents on ma- chine learning experimentation","venue":"cs.LG","work_id":"5655b20a-fbb7-4a39-8605-d6e1d689895a","year":2023},"citing_paper":{"arxiv_id":"2604.14116","last_updated":"2026-04-22T07:33:13Z","snapshot_observed_at":"2026-08-12T21:10:39.427401Z","submitted_at":"2026-04-15T17:38:06Z","title":"TREX: Automating LLM Fine-tuning via Agent-Driven Tree-based Exploration","version":2},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-05-10T12:34:29.808503Z"},"links":{"cited_paper":"/paper/2310.03302","citing_paper":"/paper/2604.14116"},"observation_digest":"sha256:4888509920a8c2a3af92c07cf3354e9b396fbfcf4bcd3f477a0e761786dc1699","observation_id":"223bdcbf-c044-43d3-a82e-c276b1d41722","resolution":{"observed_at":"2026-05-11T11:46:35.049357Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.03302","last_updated":"2024-04-14T21:02:16Z","snapshot_observed_at":"2026-08-17T14:56:37.479640Z","submitted_at":"2023-10-05T04:06:12Z","title":"MLAgentBench: Evaluating Language Agents on Machine Learning Experimentation","version":2},"cited_work":{"arxiv_id":"2310.03302","doi":null,"metadata_source":"pith","pith_arxiv_id":"2310.03302","snapshot_observed_at":"2026-07-05T17:51:14.743354Z","title":"Mlagentbench: Evaluating language agents on ma- chine learning experimentation","venue":"cs.LG","work_id":"5655b20a-fbb7-4a39-8605-d6e1d689895a","year":2023},"citing_paper":{"arxiv_id":"2604.16321","last_updated":"2026-02-25T07:55:49Z","snapshot_observed_at":"2026-08-15T23:44:14.482796Z","submitted_at":"2026-02-25T07:55:49Z","title":"LLM-Based Multi-Agent Systems for Code Generation: A Multi-Vocal Literature Review","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-05-15T19:52:49.324500Z"},"links":{"cited_paper":"/paper/2310.03302","citing_paper":"/paper/2604.16321"},"observation_digest":"sha256:18cf0bfd07f5364677f97bc49f36f1819db6883022e6ce3fc556c1c39564f061","observation_id":"8e9faee6-5976-4c31-b9fa-35f042e7b416","resolution":{"observed_at":"2026-05-15T19:56:33.865350Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.03302","last_updated":"2024-04-14T21:02:16Z","snapshot_observed_at":"2026-08-17T14:56:37.479640Z","submitted_at":"2023-10-05T04:06:12Z","title":"MLAgentBench: Evaluating Language Agents on Machine Learning Experimentation","version":2},"cited_work":{"arxiv_id":"2310.03302","doi":null,"metadata_source":"pith","pith_arxiv_id":"2310.03302","snapshot_observed_at":"2026-07-05T17:51:14.743354Z","title":"Mlagentbench: Evaluating language agents on ma- chine learning experimentation","venue":"cs.LG","work_id":"5655b20a-fbb7-4a39-8605-d6e1d689895a","year":2023},"citing_paper":{"arxiv_id":"2604.17406","last_updated":"2026-07-01T09:10:53Z","snapshot_observed_at":"2026-08-16T12:13:20.421267Z","submitted_at":"2026-04-19T12:26:05Z","title":"EvoMaster: A Foundational Evolving Agent Framework for Agentic Science at Scale","version":2},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-05-10T05:59:01.010437Z"},"links":{"cited_paper":"/paper/2310.03302","citing_paper":"/paper/2604.17406"},"observation_digest":"sha256:cc08be947c94fb271a8928da0c697072191d19c8f87e380f7cbf9d56feba322e","observation_id":"147f34cf-774e-4056-a5f2-7dbc53eb5c5c","resolution":{"observed_at":"2026-05-10T06:01:13.385053Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.03302","last_updated":"2024-04-14T21:02:16Z","snapshot_observed_at":"2026-08-17T14:56:37.479640Z","submitted_at":"2023-10-05T04:06:12Z","title":"MLAgentBench: Evaluating Language Agents on Machine Learning Experimentation","version":2},"cited_work":{"arxiv_id":"2310.03302","doi":null,"metadata_source":"pith","pith_arxiv_id":"2310.03302","snapshot_observed_at":"2026-07-05T17:51:14.743354Z","title":"Mlagentbench: Evaluating language agents on ma- chine learning experimentation","venue":"cs.LG","work_id":"5655b20a-fbb7-4a39-8605-d6e1d689895a","year":2023},"citing_paper":{"arxiv_id":"2604.17406","last_updated":"2026-07-01T09:10:53Z","snapshot_observed_at":"2026-08-16T12:13:20.421267Z","submitted_at":"2026-04-19T12:26:05Z","title":"EvoMaster: A Foundational Evolving Agent Framework for Agentic Science at Scale","version":4},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-07-05T17:45:55.631459Z"},"links":{"cited_paper":"/paper/2310.03302","citing_paper":"/paper/2604.17406"},"observation_digest":"sha256:5238891d4cea959ad387cbd8b92809a9997068f6eb6bb0fee463ecc6345a48d6","observation_id":"086c5d67-6e5b-4777-889d-4c4053514e26","resolution":{"observed_at":"2026-07-05T17:51:14.745073Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.03302","last_updated":"2024-04-14T21:02:16Z","snapshot_observed_at":"2026-08-17T14:56:37.479640Z","submitted_at":"2023-10-05T04:06:12Z","title":"MLAgentBench: Evaluating Language Agents on Machine Learning Experimentation","version":2},"cited_work":{"arxiv_id":"2310.03302","doi":null,"metadata_source":"pith","pith_arxiv_id":"2310.03302","snapshot_observed_at":"2026-07-05T17:51:14.743354Z","title":"Mlagentbench: Evaluating language agents on ma- chine learning experimentation","venue":"cs.LG","work_id":"5655b20a-fbb7-4a39-8605-d6e1d689895a","year":2023},"citing_paper":{"arxiv_id":"2605.11532","last_updated":"2026-05-12T04:58:08Z","snapshot_observed_at":"2026-08-16T11:14:23.860672Z","submitted_at":"2026-05-12T04:58:08Z","title":"Read, Grep, and Synthesize: Diagnosing Cross-Domain Seed Exposure for LLM Research Ideation","version":1},"reference_index":11,"source":"arxiv_source","source_observed_at":"2026-05-13T01:06:02.275470Z"},"links":{"cited_paper":"/paper/2310.03302","citing_paper":"/paper/2605.11532"},"observation_digest":"sha256:0edb852843cea21e303abd7c6a04e663cc19f6cb9fe46bfe957eabb28e3cbf6b","observation_id":"9faebeff-3442-4c9e-920a-b387b6f37505","resolution":{"observed_at":"2026-05-13T01:07:00.071290Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.03302","last_updated":"2024-04-14T21:02:16Z","snapshot_observed_at":"2026-08-17T14:56:37.479640Z","submitted_at":"2023-10-05T04:06:12Z","title":"MLAgentBench: Evaluating Language Agents on Machine Learning Experimentation","version":2},"cited_work":{"arxiv_id":"2310.03302","doi":null,"metadata_source":"pith","pith_arxiv_id":"2310.03302","snapshot_observed_at":"2026-07-05T17:51:14.743354Z","title":"Mlagentbench: Evaluating language agents on ma- chine learning experimentation","venue":"cs.LG","work_id":"5655b20a-fbb7-4a39-8605-d6e1d689895a","year":2023},"citing_paper":{"arxiv_id":"2605.12808","last_updated":"2026-05-14T01:55:46Z","snapshot_observed_at":"2026-08-14T06:40:13.670571Z","submitted_at":"2026-05-12T23:00:18Z","title":"Neurodata Without Boredom: Benchmarking Agentic AI for Data Reuse","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-05-14T20:23:34.514071Z"},"links":{"cited_paper":"/paper/2310.03302","citing_paper":"/paper/2605.12808"},"observation_digest":"sha256:8c778b36f0fb85fec482eaf3250b85fbacc3fe6a0b97792037f8e52b5aa8ba71","observation_id":"83d892e4-5187-48a9-84a4-4ec6ee350ed5","resolution":{"observed_at":"2026-05-14T20:42:59.121140Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.03302","last_updated":"2024-04-14T21:02:16Z","snapshot_observed_at":"2026-08-17T14:56:37.479640Z","submitted_at":"2023-10-05T04:06:12Z","title":"MLAgentBench: Evaluating Language Agents on Machine Learning Experimentation","version":2},"cited_work":{"arxiv_id":"2310.03302","doi":null,"metadata_source":"pith","pith_arxiv_id":"2310.03302","snapshot_observed_at":"2026-07-05T17:51:14.743354Z","title":"Mlagentbench: Evaluating language agents on ma- chine learning experimentation","venue":"cs.LG","work_id":"5655b20a-fbb7-4a39-8605-d6e1d689895a","year":2023},"citing_paper":{"arxiv_id":"2605.12808","last_updated":"2026-05-14T01:55:46Z","snapshot_observed_at":"2026-08-14T06:40:13.670571Z","submitted_at":"2026-05-12T23:00:18Z","title":"Neurodata Without Boredom: Benchmarking Agentic AI for Data Reuse","version":2},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-05-15T04:51:17.519200Z"},"links":{"cited_paper":"/paper/2310.03302","citing_paper":"/paper/2605.12808"},"observation_digest":"sha256:fe2f6f75ce1e6eaa55733f64a54680bb8e61d25548d31ec70485e3d1b0f69003","observation_id":"72e9c313-cd13-4bd5-a23a-2b5b36a1f247","resolution":{"observed_at":"2026-05-15T04:55:04.193134Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.03302","last_updated":"2024-04-14T21:02:16Z","snapshot_observed_at":"2026-08-17T14:56:37.479640Z","submitted_at":"2023-10-05T04:06:12Z","title":"MLAgentBench: Evaluating Language Agents on Machine Learning Experimentation","version":2},"cited_work":{"arxiv_id":"2310.03302","doi":null,"metadata_source":"pith","pith_arxiv_id":"2310.03302","snapshot_observed_at":"2026-07-05T17:51:14.743354Z","title":"Mlagentbench: Evaluating language agents on ma- chine learning experimentation","venue":"cs.LG","work_id":"5655b20a-fbb7-4a39-8605-d6e1d689895a","year":2023},"citing_paper":{"arxiv_id":"2605.13950","last_updated":"2026-05-13T18:00:00Z","snapshot_observed_at":"2026-08-13T15:32:30.494754Z","submitted_at":"2026-05-13T18:00:00Z","title":"Collider-Bench: Benchmarking AI Agents with Particle Physics Analysis Reproduction","version":1},"reference_index":20,"source":"arxiv_source","source_observed_at":"2026-05-15T06:04:03.605898Z"},"links":{"cited_paper":"/paper/2310.03302","citing_paper":"/paper/2605.13950"},"observation_digest":"sha256:ac9806cad38ce0877280a5e29a3ef85fa6f4e0e9bb7bd39d6af8d366a33836fe","observation_id":"6ab426c4-9ec5-48d3-b3f7-f1bbf567ca0d","resolution":{"observed_at":"2026-05-15T06:05:06.747450Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.03302","last_updated":"2024-04-14T21:02:16Z","snapshot_observed_at":"2026-08-17T14:56:37.479640Z","submitted_at":"2023-10-05T04:06:12Z","title":"MLAgentBench: Evaluating Language Agents on Machine Learning Experimentation","version":2},"cited_work":{"arxiv_id":"2310.03302","doi":null,"metadata_source":"pith","pith_arxiv_id":"2310.03302","snapshot_observed_at":"2026-07-05T17:51:14.743354Z","title":"Mlagentbench: Evaluating language agents on ma- chine learning experimentation","venue":"cs.LG","work_id":"5655b20a-fbb7-4a39-8605-d6e1d689895a","year":2023},"citing_paper":{"arxiv_id":"2605.15766","last_updated":"2026-05-15T09:24:55Z","snapshot_observed_at":"2026-08-16T02:56:23.468014Z","submitted_at":"2026-05-15T09:24:55Z","title":"BioXArena: Benchmarking LLM Agents on Multi-Modal Biomedical Machine Learning Tasks","version":1},"reference_index":45,"source":"pdf_text","source_observed_at":"2026-05-19T19:31:32.334837Z"},"links":{"cited_paper":"/paper/2310.03302","citing_paper":"/paper/2605.15766"},"observation_digest":"sha256:ba8a1813ef9cceb3f680a0ac662f732d216e4642c85ef03ea071db7e74680528","observation_id":"7d92aaab-5869-465d-b174-a9f8e419e1e8","resolution":{"observed_at":"2026-05-19T19:32:43.734709Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.03302","last_updated":"2024-04-14T21:02:16Z","snapshot_observed_at":"2026-08-17T14:56:37.479640Z","submitted_at":"2023-10-05T04:06:12Z","title":"MLAgentBench: Evaluating Language Agents on Machine Learning Experimentation","version":2},"cited_work":{"arxiv_id":"2310.03302","doi":null,"metadata_source":"pith","pith_arxiv_id":"2310.03302","snapshot_observed_at":"2026-07-05T17:51:14.743354Z","title":"Mlagentbench: Evaluating language agents on ma- chine learning experimentation","venue":"cs.LG","work_id":"5655b20a-fbb7-4a39-8605-d6e1d689895a","year":2023},"citing_paper":{"arxiv_id":"2605.16616","last_updated":"2026-05-15T20:35:32Z","snapshot_observed_at":"2026-08-16T06:46:32.334621Z","submitted_at":"2026-05-15T20:35:32Z","title":"MLReplicate: Benchmarking Autonomous Research Systems for Machine Learning Reproducibility","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-05-20T19:59:40.519962Z"},"links":{"cited_paper":"/paper/2310.03302","citing_paper":"/paper/2605.16616"},"observation_digest":"sha256:1695ee27e62601bbe7c6e7f64a8c80073cc2db18dcb73cf8b1d8555daa8f94a9","observation_id":"faa7c95a-d138-42c4-95fb-b654ee60a4e1","resolution":{"observed_at":"2026-05-20T20:03:43.962049Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.03302","last_updated":"2024-04-14T21:02:16Z","snapshot_observed_at":"2026-08-17T14:56:37.479640Z","submitted_at":"2023-10-05T04:06:12Z","title":"MLAgentBench: Evaluating Language Agents on Machine Learning Experimentation","version":2},"cited_work":{"arxiv_id":"2310.03302","doi":null,"metadata_source":"pith","pith_arxiv_id":"2310.03302","snapshot_observed_at":"2026-07-05T17:51:14.743354Z","title":"Mlagentbench: Evaluating language agents on ma- chine learning experimentation","venue":"cs.LG","work_id":"5655b20a-fbb7-4a39-8605-d6e1d689895a","year":2023},"citing_paper":{"arxiv_id":"2605.17373","last_updated":"2026-05-29T04:24:32Z","snapshot_observed_at":"2026-08-15T09:26:02.151681Z","submitted_at":"2026-05-17T10:30:38Z","title":"FML-bench: A Controlled Study of AI Research Agent Strategies from the Perspective of Search Dynamics","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-05-20T14:25:15.565386Z"},"links":{"cited_paper":"/paper/2310.03302","citing_paper":"/paper/2605.17373"},"observation_digest":"sha256:a61c01b585522caa759631d9fc6895e53fe0e7c2856ccec06bbe6d20ce6c3caf","observation_id":"d0dab6a4-af03-438e-b8d0-050a0e77b66f","resolution":{"observed_at":"2026-05-20T14:28:21.467445Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.03302","last_updated":"2024-04-14T21:02:16Z","snapshot_observed_at":"2026-08-17T14:56:37.479640Z","submitted_at":"2023-10-05T04:06:12Z","title":"MLAgentBench: Evaluating Language Agents on Machine Learning Experimentation","version":2},"cited_work":{"arxiv_id":"2310.03302","doi":null,"metadata_source":"pith","pith_arxiv_id":"2310.03302","snapshot_observed_at":"2026-07-05T17:51:14.743354Z","title":"Mlagentbench: Evaluating language agents on ma- chine learning experimentation","venue":"cs.LG","work_id":"5655b20a-fbb7-4a39-8605-d6e1d689895a","year":2023},"citing_paper":{"arxiv_id":"2605.17373","last_updated":"2026-05-29T04:24:32Z","snapshot_observed_at":"2026-08-15T09:26:02.151681Z","submitted_at":"2026-05-17T10:30:38Z","title":"FML-bench: A Controlled Study of AI Research Agent Strategies from the Perspective of Search Dynamics","version":2},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-06-30T19:00:30.961402Z"},"links":{"cited_paper":"/paper/2310.03302","citing_paper":"/paper/2605.17373"},"observation_digest":"sha256:eeec9de13beacb04aae2abff0c54cbc93436a37846de6a3f4a9f5a96ee911428","observation_id":"65df9179-0ebe-4a8f-acc9-3fa4c16117b7","resolution":{"observed_at":"2026-06-30T19:05:00.912674Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.03302","last_updated":"2024-04-14T21:02:16Z","snapshot_observed_at":"2026-08-17T14:56:37.479640Z","submitted_at":"2023-10-05T04:06:12Z","title":"MLAgentBench: Evaluating Language Agents on Machine Learning Experimentation","version":2},"cited_work":{"arxiv_id":"2310.03302","doi":null,"metadata_source":"pith","pith_arxiv_id":"2310.03302","snapshot_observed_at":"2026-07-05T17:51:14.743354Z","title":"Mlagentbench: Evaluating language agents on ma- chine learning experimentation","venue":"cs.LG","work_id":"5655b20a-fbb7-4a39-8605-d6e1d689895a","year":2023},"citing_paper":{"arxiv_id":"2605.19156","last_updated":"2026-05-18T22:20:33Z","snapshot_observed_at":"2026-08-15T00:50:05.932495Z","submitted_at":"2026-05-18T22:20:33Z","title":"How Far Are We From True Auto-Research?","version":1},"reference_index":21,"source":"arxiv_source","source_observed_at":"2026-05-20T09:56:16.160551Z"},"links":{"cited_paper":"/paper/2310.03302","citing_paper":"/paper/2605.19156"},"observation_digest":"sha256:75616c648741c5f1c01f030f6f432c40e118dc8c0d9498a9667a8f075b0831aa","observation_id":"9d3d2142-eb7d-4ff4-8b8b-ced9c26b255a","resolution":{"observed_at":"2026-05-20T09:58:11.238700Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.03302","last_updated":"2024-04-14T21:02:16Z","snapshot_observed_at":"2026-08-17T14:56:37.479640Z","submitted_at":"2023-10-05T04:06:12Z","title":"MLAgentBench: Evaluating Language Agents on Machine Learning Experimentation","version":2},"cited_work":{"arxiv_id":"2310.03302","doi":null,"metadata_source":"pith","pith_arxiv_id":"2310.03302","snapshot_observed_at":"2026-07-05T17:51:14.743354Z","title":"Mlagentbench: Evaluating language agents on ma- chine learning experimentation","venue":"cs.LG","work_id":"5655b20a-fbb7-4a39-8605-d6e1d689895a","year":2023},"citing_paper":{"arxiv_id":"2605.23204","last_updated":"2026-05-22T03:40:30Z","snapshot_observed_at":"2026-07-06T23:33:29.550551Z","submitted_at":"2026-05-22T03:40:30Z","title":"AutoResearch AI: Towards AI-Powered Research Automation for Scientific Discovery","version":1},"reference_index":119,"source":"pdf_text","source_observed_at":"2026-05-25T04:46:43.679185Z"},"links":{"cited_paper":"/paper/2310.03302","citing_paper":"/paper/2605.23204"},"observation_digest":"sha256:164ecac0cadfced49fde79792fb3d952254d72759fb69dcc4d66a714f583cea7","observation_id":"792e23f6-c3ff-4f18-b586-b2a53000ffce","resolution":{"observed_at":"2026-05-25T04:50:21.832940Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.03302","last_updated":"2024-04-14T21:02:16Z","snapshot_observed_at":"2026-08-17T14:56:37.479640Z","submitted_at":"2023-10-05T04:06:12Z","title":"MLAgentBench: Evaluating Language Agents on Machine Learning Experimentation","version":2},"cited_work":{"arxiv_id":"2310.03302","doi":null,"metadata_source":"pith","pith_arxiv_id":"2310.03302","snapshot_observed_at":"2026-07-05T17:51:14.743354Z","title":"Mlagentbench: Evaluating language agents on ma- chine learning experimentation","venue":"cs.LG","work_id":"5655b20a-fbb7-4a39-8605-d6e1d689895a","year":2023},"citing_paper":{"arxiv_id":"2605.26340","last_updated":"2026-05-25T21:30:27Z","snapshot_observed_at":"2026-08-14T19:28:55.309579Z","submitted_at":"2026-05-25T21:30:27Z","title":"ScientistOne: Towards Human-Level Autonomous Research via Chain-of-Evidence","version":1},"reference_index":8,"source":"arxiv_source","source_observed_at":"2026-06-29T21:19:03.281629Z"},"links":{"cited_paper":"/paper/2310.03302","citing_paper":"/paper/2605.26340"},"observation_digest":"sha256:9e3037f5c92625b59c23b60569ad1e1f3c8b742973aa3f59a11e9d5285c9fc05","observation_id":"f42692fa-8b91-4298-acb2-9a463988e008","resolution":{"observed_at":"2026-06-29T21:23:59.113747Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.03302","last_updated":"2024-04-14T21:02:16Z","snapshot_observed_at":"2026-08-17T14:56:37.479640Z","submitted_at":"2023-10-05T04:06:12Z","title":"MLAgentBench: Evaluating Language Agents on Machine Learning Experimentation","version":2},"cited_work":{"arxiv_id":"2310.03302","doi":null,"metadata_source":"pith","pith_arxiv_id":"2310.03302","snapshot_observed_at":"2026-07-05T17:51:14.743354Z","title":"Mlagentbench: Evaluating language agents on ma- chine learning experimentation","venue":"cs.LG","work_id":"5655b20a-fbb7-4a39-8605-d6e1d689895a","year":2023},"citing_paper":{"arxiv_id":"2605.28655","last_updated":"2026-05-27T15:56:12Z","snapshot_observed_at":"2026-08-12T13:47:46.993153Z","submitted_at":"2026-05-27T15:56:12Z","title":"AutoScientists: Self-Organizing Agent Teams for Long-Running Scientific Experimentation","version":1},"reference_index":54,"source":"pdf_text","source_observed_at":"2026-06-29T12:39:00.460842Z"},"links":{"cited_paper":"/paper/2310.03302","citing_paper":"/paper/2605.28655"},"observation_digest":"sha256:c4ab2d7257326b50bdbef6f7db6dc659b315c15fbbf223947a9d08665303da11","observation_id":"10091f2b-0aad-42d7-bf17-f83624152935","resolution":{"observed_at":"2026-06-29T12:43:25.675420Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.03302","last_updated":"2024-04-14T21:02:16Z","snapshot_observed_at":"2026-08-17T14:56:37.479640Z","submitted_at":"2023-10-05T04:06:12Z","title":"MLAgentBench: Evaluating Language Agents on Machine Learning Experimentation","version":2},"cited_work":{"arxiv_id":"2310.03302","doi":null,"metadata_source":"pith","pith_arxiv_id":"2310.03302","snapshot_observed_at":"2026-07-05T17:51:14.743354Z","title":"Mlagentbench: Evaluating language agents on ma- chine learning experimentation","venue":"cs.LG","work_id":"5655b20a-fbb7-4a39-8605-d6e1d689895a","year":2023},"citing_paper":{"arxiv_id":"2605.30434","last_updated":"2026-05-28T18:00:20Z","snapshot_observed_at":"2026-08-16T08:14:10.820776Z","submitted_at":"2026-05-28T18:00:20Z","title":"LongDS-Bench: On the Failure of Long-Horizon Agentic Data Analysis","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-06-29T08:56:24.657380Z"},"links":{"cited_paper":"/paper/2310.03302","citing_paper":"/paper/2605.30434"},"observation_digest":"sha256:381fba8121178987b240d15058b88e8f3e0e7fbd2a83e8eb161fc194b0bba1f6","observation_id":"adf244bb-13de-4a2d-be08-e6597990a58a","resolution":{"observed_at":"2026-06-29T09:03:16.040070Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.03302","last_updated":"2024-04-14T21:02:16Z","snapshot_observed_at":"2026-08-17T14:56:37.479640Z","submitted_at":"2023-10-05T04:06:12Z","title":"MLAgentBench: Evaluating Language Agents on Machine Learning Experimentation","version":2},"cited_work":{"arxiv_id":"2310.03302","doi":null,"metadata_source":"pith","pith_arxiv_id":"2310.03302","snapshot_observed_at":"2026-07-05T17:51:14.743354Z","title":"Mlagentbench: Evaluating language agents on ma- chine learning experimentation","venue":"cs.LG","work_id":"5655b20a-fbb7-4a39-8605-d6e1d689895a","year":2023},"citing_paper":{"arxiv_id":"2606.04261","last_updated":"2026-06-02T22:26:53Z","snapshot_observed_at":"2026-08-14T12:48:45.713641Z","submitted_at":"2026-06-02T22:26:53Z","title":"Can Generalist Agents Automate Data Curation?","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-06-28T09:32:39.361415Z"},"links":{"cited_paper":"/paper/2310.03302","citing_paper":"/paper/2606.04261"},"observation_digest":"sha256:af23679b4a38848e87bb1389409051bab6e85f1aaf3c6d8684358aa0c2152b28","observation_id":"a251f68d-354f-4671-9add-bf046b693955","resolution":{"observed_at":"2026-07-02T03:56:35.140323Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.03302","last_updated":"2024-04-14T21:02:16Z","snapshot_observed_at":"2026-08-17T14:56:37.479640Z","submitted_at":"2023-10-05T04:06:12Z","title":"MLAgentBench: Evaluating Language Agents on Machine Learning Experimentation","version":2},"cited_work":{"arxiv_id":"2310.03302","doi":null,"metadata_source":"pith","pith_arxiv_id":"2310.03302","snapshot_observed_at":"2026-07-05T17:51:14.743354Z","title":"Mlagentbench: Evaluating language agents on ma- chine learning experimentation","venue":"cs.LG","work_id":"5655b20a-fbb7-4a39-8605-d6e1d689895a","year":2023},"citing_paper":{"arxiv_id":"2606.05250","last_updated":"2026-06-03T12:56:11Z","snapshot_observed_at":"2026-08-14T04:55:08.035683Z","submitted_at":"2026-06-03T12:56:11Z","title":"Towards Persistent Case-Based Memory for Autonomous Data Science: A CBR-Augmented R&D-Agent with a Locally Deployable Small Language Model","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-06-28T05:19:14.975753Z"},"links":{"cited_paper":"/paper/2310.03302","citing_paper":"/paper/2606.05250"},"observation_digest":"sha256:c64afb54adcc33555ff26924d0804054148a6a1c6ecb6e10fb1deff9f3695e4e","observation_id":"137e84eb-005c-41d3-b6ad-e2b7e31e2fdf","resolution":{"observed_at":"2026-07-02T10:06:51.697112Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.03302","last_updated":"2024-04-14T21:02:16Z","snapshot_observed_at":"2026-08-17T14:56:37.479640Z","submitted_at":"2023-10-05T04:06:12Z","title":"MLAgentBench: Evaluating Language Agents on Machine Learning Experimentation","version":2},"cited_work":{"arxiv_id":"2310.03302","doi":null,"metadata_source":"pith","pith_arxiv_id":"2310.03302","snapshot_observed_at":"2026-07-05T17:51:14.743354Z","title":"Mlagentbench: Evaluating language agents on ma- chine learning experimentation","venue":"cs.LG","work_id":"5655b20a-fbb7-4a39-8605-d6e1d689895a","year":2023},"citing_paper":{"arxiv_id":"2606.06473","last_updated":"2026-06-04T17:55:59Z","snapshot_observed_at":"2026-08-14T17:51:52.839491Z","submitted_at":"2026-06-04T17:55:59Z","title":"MLEvolve: A Self-Evolving Framework for Automated Machine Learning Algorithm Discovery","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-06-28T00:57:02.959467Z"},"links":{"cited_paper":"/paper/2310.03302","citing_paper":"/paper/2606.06473"},"observation_digest":"sha256:0f756e9919599dbd5f50fe73a199ad5dca85f9458285c976c1222ed71dde1bdb","observation_id":"2e5178ee-23cf-4342-8f27-b7e1bedac843","resolution":{"observed_at":"2026-07-02T13:46:59.786067Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.03302","last_updated":"2024-04-14T21:02:16Z","snapshot_observed_at":"2026-08-17T14:56:37.479640Z","submitted_at":"2023-10-05T04:06:12Z","title":"MLAgentBench: Evaluating Language Agents on Machine Learning Experimentation","version":2},"cited_work":{"arxiv_id":"2310.03302","doi":null,"metadata_source":"pith","pith_arxiv_id":"2310.03302","snapshot_observed_at":"2026-07-05T17:51:14.743354Z","title":"Mlagentbench: Evaluating language agents on ma- chine learning experimentation","venue":"cs.LG","work_id":"5655b20a-fbb7-4a39-8605-d6e1d689895a","year":2023},"citing_paper":{"arxiv_id":"2606.07591","last_updated":"2026-07-03T01:40:25Z","snapshot_observed_at":"2026-08-14T09:41:17.327665Z","submitted_at":"2026-05-28T16:27:40Z","title":"ResearchClawBench: A Benchmark for End-to-End Autonomous Scientific Research","version":3},"reference_index":10,"source":"arxiv_source","source_observed_at":"2026-06-29T08:24:42.412763Z"},"links":{"cited_paper":"/paper/2310.03302","citing_paper":"/paper/2606.07591"},"observation_digest":"sha256:5b788b68a40ae24a3165def3e921b0791be1cde9d5317945520e60d02961ae27","observation_id":"15c51748-9d40-4cd7-bb6b-27f5de7401c7","resolution":{"observed_at":"2026-06-29T08:33:15.805420Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.03302","last_updated":"2024-04-14T21:02:16Z","snapshot_observed_at":"2026-08-17T14:56:37.479640Z","submitted_at":"2023-10-05T04:06:12Z","title":"MLAgentBench: Evaluating Language Agents on Machine Learning Experimentation","version":2},"cited_work":{"arxiv_id":"2310.03302","doi":null,"metadata_source":"pith","pith_arxiv_id":"2310.03302","snapshot_observed_at":"2026-07-05T17:51:14.743354Z","title":"Mlagentbench: Evaluating language agents on ma- chine learning experimentation","venue":"cs.LG","work_id":"5655b20a-fbb7-4a39-8605-d6e1d689895a","year":2023},"citing_paper":{"arxiv_id":"2606.07591","last_updated":"2026-07-03T01:40:25Z","snapshot_observed_at":"2026-08-14T09:41:17.327665Z","submitted_at":"2026-05-28T16:27:40Z","title":"ResearchClawBench: A Benchmark for End-to-End Autonomous Scientific Research","version":4},"reference_index":10,"source":"arxiv_source","source_observed_at":"2026-07-04T00:30:00.449270Z"},"links":{"cited_paper":"/paper/2310.03302","citing_paper":"/paper/2606.07591"},"observation_digest":"sha256:9085bb210e2778fb5a6974aab895482320446efafefe6401efdb662c999354b5","observation_id":"0f774748-35f1-4c5d-8cef-ef19ab480933","resolution":{"observed_at":"2026-07-04T00:39:16.575746Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.03302","last_updated":"2024-04-14T21:02:16Z","snapshot_observed_at":"2026-08-17T14:56:37.479640Z","submitted_at":"2023-10-05T04:06:12Z","title":"MLAgentBench: Evaluating Language Agents on Machine Learning Experimentation","version":2},"cited_work":{"arxiv_id":"2310.03302","doi":null,"metadata_source":"pith","pith_arxiv_id":"2310.03302","snapshot_observed_at":"2026-07-05T17:51:14.743354Z","title":"Mlagentbench: Evaluating language agents on ma- chine learning experimentation","venue":"cs.LG","work_id":"5655b20a-fbb7-4a39-8605-d6e1d689895a","year":2023},"citing_paper":{"arxiv_id":"2606.11176","last_updated":"2026-06-09T17:51:55Z","snapshot_observed_at":"2026-08-04T10:02:13.493460Z","submitted_at":"2026-06-09T17:51:55Z","title":"Data Journalist Agent: Transforming Data into Verifiable Multimodal Stories","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-06-27T13:43:02.919248Z"},"links":{"cited_paper":"/paper/2310.03302","citing_paper":"/paper/2606.11176"},"observation_digest":"sha256:8493e9aaf53727d9e1955861fa461d2ea6f56efa3653c4f1f7a35226dd453433","observation_id":"701e14d2-be9f-4733-a559-bf63a93985bb","resolution":{"observed_at":"2026-07-03T04:37:37.772034Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.03302","last_updated":"2024-04-14T21:02:16Z","snapshot_observed_at":"2026-08-17T14:56:37.479640Z","submitted_at":"2023-10-05T04:06:12Z","title":"MLAgentBench: Evaluating Language Agents on Machine Learning Experimentation","version":2},"cited_work":{"arxiv_id":"2310.03302","doi":null,"metadata_source":"pith","pith_arxiv_id":"2310.03302","snapshot_observed_at":"2026-07-05T17:51:14.743354Z","title":"Mlagentbench: Evaluating language agents on ma- chine learning experimentation","venue":"cs.LG","work_id":"5655b20a-fbb7-4a39-8605-d6e1d689895a","year":2023},"citing_paper":{"arxiv_id":"2606.17915","last_updated":"2026-06-16T13:34:27Z","snapshot_observed_at":"2026-07-06T23:53:25.839293Z","submitted_at":"2026-06-16T13:34:27Z","title":"Trustworthy Self-Composable Big-Data-as-a-Service: An LLM-Orchestrated Multi-Agent Framework for Automated Data Engineering, AutoML, MLOps Deployment, and Drift-Aware Lifecycle Optimization","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-06-26T22:07:45.909694Z"},"links":{"cited_paper":"/paper/2310.03302","citing_paper":"/paper/2606.17915"},"observation_digest":"sha256:a882cb424954caf3db38e1528376cec14a860aaf0d86866f2fe4db83c3b27b54","observation_id":"84420d36-0c62-49d9-9907-497a96646daf","resolution":{"observed_at":"2026-07-03T23:29:02.943758Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.03302","last_updated":"2024-04-14T21:02:16Z","snapshot_observed_at":"2026-08-17T14:56:37.479640Z","submitted_at":"2023-10-05T04:06:12Z","title":"MLAgentBench: Evaluating Language Agents on Machine Learning Experimentation","version":2},"cited_work":{"arxiv_id":"2310.03302","doi":null,"metadata_source":"pith","pith_arxiv_id":"2310.03302","snapshot_observed_at":"2026-07-05T17:51:14.743354Z","title":"Mlagentbench: Evaluating language agents on ma- chine learning experimentation","venue":"cs.LG","work_id":"5655b20a-fbb7-4a39-8605-d6e1d689895a","year":2023},"citing_paper":{"arxiv_id":"2606.20394","last_updated":"2026-06-18T15:48:25Z","snapshot_observed_at":"2026-08-16T02:57:57.988956Z","submitted_at":"2026-06-18T15:48:25Z","title":"Agentic AutoResearch forSpace Autonomy: An Auditable, LLM-Driven Research Agent for Aerospace Control Problems","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-06-26T17:29:59.914576Z"},"links":{"cited_paper":"/paper/2310.03302","citing_paper":"/paper/2606.20394"},"observation_digest":"sha256:8dc3d71cf78f4f9346018ab7173fbf0bee06fa984fed4502a644139f4b21f4f1","observation_id":"9cd2f965-2ec6-4874-b5dd-4df2b11b08cd","resolution":{"observed_at":"2026-07-04T03:59:32.715993Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.03302","last_updated":"2024-04-14T21:02:16Z","snapshot_observed_at":"2026-08-17T14:56:37.479640Z","submitted_at":"2023-10-05T04:06:12Z","title":"MLAgentBench: Evaluating Language Agents on Machine Learning Experimentation","version":2},"cited_work":{"arxiv_id":"2310.03302","doi":null,"metadata_source":"pith","pith_arxiv_id":"2310.03302","snapshot_observed_at":"2026-07-05T17:51:14.743354Z","title":"Mlagentbench: Evaluating language agents on ma- chine learning experimentation","venue":"cs.LG","work_id":"5655b20a-fbb7-4a39-8605-d6e1d689895a","year":2023},"citing_paper":{"arxiv_id":"2606.22610","last_updated":"2026-06-21T17:37:01Z","snapshot_observed_at":"2026-07-06T23:57:28.172988Z","submitted_at":"2026-06-21T17:37:01Z","title":"PaperClaw: Harnessing Agents for Autonomous Research and Human-in-the-Loop Refinement","version":1},"reference_index":43,"source":"arxiv_source","source_observed_at":"2026-06-26T10:36:14.840581Z"},"links":{"cited_paper":"/paper/2310.03302","citing_paper":"/paper/2606.22610"},"observation_digest":"sha256:fd77308c5b7d411de38e02d8b0a405fc21bd76ecf941b0d0f489d9939934250e","observation_id":"a9e68e22-d9e6-4cc4-86ef-bfa3f0a25ec7","resolution":{"observed_at":"2026-07-04T08:59:43.504947Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.03302","last_updated":"2024-04-14T21:02:16Z","snapshot_observed_at":"2026-08-17T14:56:37.479640Z","submitted_at":"2023-10-05T04:06:12Z","title":"MLAgentBench: Evaluating Language Agents on Machine Learning Experimentation","version":2},"cited_work":{"arxiv_id":"2310.03302","doi":null,"metadata_source":"pith","pith_arxiv_id":"2310.03302","snapshot_observed_at":"2026-07-05T17:51:14.743354Z","title":"Mlagentbench: Evaluating language agents on ma- chine learning experimentation","venue":"cs.LG","work_id":"5655b20a-fbb7-4a39-8605-d6e1d689895a","year":2023},"citing_paper":{"arxiv_id":"2606.24530","last_updated":"2026-07-06T16:56:53Z","snapshot_observed_at":"2026-08-15T20:31:56.652908Z","submitted_at":"2026-06-23T12:58:23Z","title":"NatureBench: Can Coding Agents Match the Published SOTA of Nature-Family Papers?","version":1},"reference_index":33,"source":"arxiv_source","source_observed_at":"2026-06-26T00:13:14.940915Z"},"links":{"cited_paper":"/paper/2310.03302","citing_paper":"/paper/2606.24530"},"observation_digest":"sha256:94b2c099291fdda23828583d90f9b6f69eaab03fa89284a127a8b3c5503fa65e","observation_id":"f90db1d0-05fc-4791-aa6b-ce31a1fafd1c","resolution":{"observed_at":"2026-07-04T16:49:57.799757Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.03302","last_updated":"2024-04-14T21:02:16Z","snapshot_observed_at":"2026-08-17T14:56:37.479640Z","submitted_at":"2023-10-05T04:06:12Z","title":"MLAgentBench: Evaluating Language Agents on Machine Learning Experimentation","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.03302","snapshot_observed_at":"2026-07-12T12:32:39.706055Z","title":"MLAgentBench : Evaluating language agents on machine learning experimentation, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2606.24530","last_updated":"2026-07-06T16:56:53Z","snapshot_observed_at":"2026-08-15T20:31:56.652908Z","submitted_at":"2026-06-23T12:58:23Z","title":"NatureBench: Can Coding Agents Match the Published SOTA of Nature-Family Papers?","version":2},"reference_index":33,"source":"arxiv_source","source_observed_at":"2026-07-12T12:32:39.706055Z"},"links":{"cited_paper":"/paper/2310.03302","citing_paper":"/paper/2606.24530"},"observation_digest":"sha256:5540e0df6f35fef34cc92005811c47d0351d612828ec3436f76d6015b1f055dd","observation_id":"34e973d5-9efb-483a-94d1-008b7dde1658","resolution":{"observed_at":"2026-07-12T12:32:39.706055Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.03302","last_updated":"2024-04-14T21:02:16Z","snapshot_observed_at":"2026-08-17T14:56:37.479640Z","submitted_at":"2023-10-05T04:06:12Z","title":"MLAgentBench: Evaluating Language Agents on Machine Learning Experimentation","version":2},"cited_work":{"arxiv_id":"2310.03302","doi":null,"metadata_source":"pith","pith_arxiv_id":"2310.03302","snapshot_observed_at":"2026-07-05T17:51:14.743354Z","title":"Mlagentbench: Evaluating language agents on ma- chine learning experimentation","venue":"cs.LG","work_id":"5655b20a-fbb7-4a39-8605-d6e1d689895a","year":2023},"citing_paper":{"arxiv_id":"2606.27416","last_updated":"2026-06-25T17:52:26Z","snapshot_observed_at":"2026-08-02T19:52:21.146553Z","submitted_at":"2026-06-25T17:52:26Z","title":"Glite ARF: Verifier-Driven Research with Parallel LLM Coding Agents","version":1},"reference_index":46,"source":"arxiv_source","source_observed_at":"2026-06-29T01:15:00.436786Z"},"links":{"cited_paper":"/paper/2310.03302","citing_paper":"/paper/2606.27416"},"observation_digest":"sha256:d694da5056b5af4884fa7f988cdfda964b2084cc180ffcf1f9c9eecdd529e9ef","observation_id":"9c1c4257-7653-40c1-84d5-31e249cf05cf","resolution":{"observed_at":"2026-07-01T19:06:02.745752Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.03302","last_updated":"2024-04-14T21:02:16Z","snapshot_observed_at":"2026-08-17T14:56:37.479640Z","submitted_at":"2023-10-05T04:06:12Z","title":"MLAgentBench: Evaluating Language Agents on Machine Learning Experimentation","version":2},"cited_work":{"arxiv_id":"2310.03302","doi":null,"metadata_source":"pith","pith_arxiv_id":"2310.03302","snapshot_observed_at":"2026-07-05T17:51:14.743354Z","title":"Mlagentbench: Evaluating language agents on ma- chine learning experimentation","venue":"cs.LG","work_id":"5655b20a-fbb7-4a39-8605-d6e1d689895a","year":2023},"citing_paper":{"arxiv_id":"2606.29537","last_updated":"2026-07-13T10:30:17Z","snapshot_observed_at":"2026-08-14T04:58:24.703062Z","submitted_at":"2026-06-28T17:59:17Z","title":"OSWorld 2.0: Benchmarking Computer Use Agents on Long-Horizon Real-World Tasks","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-06-30T07:10:38.909339Z"},"links":{"cited_paper":"/paper/2310.03302","citing_paper":"/paper/2606.29537"},"observation_digest":"sha256:44ab3fa18734748beb2455751d7dc115c18ab40c1f1910958fd298c315b4febe","observation_id":"ebac2fff-1b08-4520-86b8-79572d0a36ec","resolution":{"observed_at":"2026-06-30T07:14:21.196292Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.03302","last_updated":"2024-04-14T21:02:16Z","snapshot_observed_at":"2026-08-17T14:56:37.479640Z","submitted_at":"2023-10-05T04:06:12Z","title":"MLAgentBench: Evaluating Language Agents on Machine Learning Experimentation","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.03302","snapshot_observed_at":"2026-07-15T10:24:53.345620Z","title":"MLAgentBench: Evaluating language agents on machine learning experimentation, 2024","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2606.29537","last_updated":"2026-07-13T10:30:17Z","snapshot_observed_at":"2026-08-14T04:58:24.703062Z","submitted_at":"2026-06-28T17:59:17Z","title":"OSWorld 2.0: Benchmarking Computer Use Agents on Long-Horizon Real-World Tasks","version":2},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-07-15T10:24:53.345620Z"},"links":{"cited_paper":"/paper/2310.03302","citing_paper":"/paper/2606.29537"},"observation_digest":"sha256:86a5621be9b71e806f2433c57e92e91c986e0292b8f9a27c6c7bdde3657440a1","observation_id":"5df1d11f-51c6-4bf4-85fa-92d80987df8b","resolution":{"observed_at":"2026-07-15T10:24:53.345620Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.03302","last_updated":"2024-04-14T21:02:16Z","snapshot_observed_at":"2026-08-17T14:56:37.479640Z","submitted_at":"2023-10-05T04:06:12Z","title":"MLAgentBench: Evaluating Language Agents on Machine Learning Experimentation","version":2},"cited_work":{"arxiv_id":"2310.03302","doi":null,"metadata_source":"pith","pith_arxiv_id":"2310.03302","snapshot_observed_at":"2026-07-05T17:51:14.743354Z","title":"Mlagentbench: Evaluating language agents on ma- chine learning experimentation","venue":"cs.LG","work_id":"5655b20a-fbb7-4a39-8605-d6e1d689895a","year":2023},"citing_paper":{"arxiv_id":"2606.31651","last_updated":"2026-07-13T07:29:01Z","snapshot_observed_at":"2026-08-17T07:21:18.082974Z","submitted_at":"2026-06-30T13:30:24Z","title":"FARS: A Fully Automated Research System Deployed at Scale","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-07-01T05:18:53.840963Z"},"links":{"cited_paper":"/paper/2310.03302","citing_paper":"/paper/2606.31651"},"observation_digest":"sha256:46ec356bfed3b82075fc5c32af4d5352a0190eaeac27a7f5700cb453b7165ca2","observation_id":"7ee6a2fc-46ff-4d74-b42a-5d17c42349a2","resolution":{"observed_at":"2026-07-01T10:35:42.748977Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-17T06:30:58.91139+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.03302","last_updated":"2024-04-14T21:02:16Z","snapshot_observed_at":"2026-08-17T14:56:37.479640Z","submitted_at":"2023-10-05T04:06:12Z","title":"MLAgentBench: Evaluating Language Agents on Machine Learning Experimentation","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.03302","snapshot_observed_at":"2026-07-14T16:55:37.417202Z","title":"Intology","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2606.31651","last_updated":"2026-07-13T07:29:01Z","snapshot_observed_at":"2026-08-17T07:21:18.082974Z","submitted_at":"2026-06-30T13:30:24Z","title":"FARS: A Fully Automated Research System Deployed at Scale","version":2},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-07-14T16:55:37.417202Z"},"links":{"cited_paper":"/paper/2310.03302","citing_paper":"/paper/2606.31651"},"observation_digest":"sha256:44e0e13f603c3d10100b5a1db2fb7648c2b42ff9a40511079400e4036be350da","observation_id":"490b25af-c307-4926-b17b-7874354c2ff5","resolution":{"observed_at":"2026-07-14T16:55:37.417202Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.03302","last_updated":"2024-04-14T21:02:16Z","snapshot_observed_at":"2026-08-17T14:56:37.479640Z","submitted_at":"2023-10-05T04:06:12Z","title":"MLAgentBench: Evaluating Language Agents on Machine Learning Experimentation","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.03302","snapshot_observed_at":"2026-07-12T05:14:41.315225Z","title":"In: International Conference on Machine Learning (2024)","venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.03048","last_updated":"2026-07-03T07:37:57Z","snapshot_observed_at":"2026-08-16T17:47:44.186685Z","submitted_at":"2026-07-03T07:37:57Z","title":"Compression, structure, and executor capability: a controlled real-cost decomposition of language-model agent skill optimisation","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-07-12T05:14:41.315225Z"},"links":{"cited_paper":"/paper/2310.03302","citing_paper":"/paper/2607.03048"},"observation_digest":"sha256:05ebcd506b514e94a7f0fb0032878f7b1407213fb7ece98e7667a222040b01f7","observation_id":"d6bea006-f3cf-4edf-8e8c-13f3d8637773","resolution":{"observed_at":"2026-07-12T05:14:41.315225Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.03302","last_updated":"2024-04-14T21:02:16Z","snapshot_observed_at":"2026-08-17T14:56:37.479640Z","submitted_at":"2023-10-05T04:06:12Z","title":"MLAgentBench: Evaluating Language Agents on Machine Learning Experimentation","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.03302","snapshot_observed_at":"2026-07-12T01:16:50.927415Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.03601","last_updated":"2026-07-03T20:58:29Z","snapshot_observed_at":"2026-08-03T18:38:21.736479Z","submitted_at":"2026-07-03T20:58:29Z","title":"ArchEval: Measuring AI Agents as Computer Architects","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-07-12T01:16:50.927415Z"},"links":{"cited_paper":"/paper/2310.03302","citing_paper":"/paper/2607.03601"},"observation_digest":"sha256:108e76ed5e4a42e144a1cc0bb29ae65d491b28379b87df45dd73a1b5f0f71783","observation_id":"5ec7eccd-f664-4f90-aedf-6adcbf74213f","resolution":{"observed_at":"2026-07-12T01:16:50.927415Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.03302","last_updated":"2024-04-14T21:02:16Z","snapshot_observed_at":"2026-08-17T14:56:37.479640Z","submitted_at":"2023-10-05T04:06:12Z","title":"MLAgentBench: Evaluating Language Agents on Machine Learning Experimentation","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.03302","snapshot_observed_at":"2026-07-14T10:46:01.272433Z","title":null,"venue":null,"work_id":null,"year":2024},"citing_paper":{"arxiv_id":"2607.10569","last_updated":"2026-07-12T04:52:08Z","snapshot_observed_at":"2026-08-14T09:00:20.788452Z","submitted_at":"2026-07-12T04:52:08Z","title":"When Does Restricting a Coding Agent to execute_code Help? A Regime $\\times$ Agent-Design Ablation","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-07-14T10:46:01.272433Z"},"links":{"cited_paper":"/paper/2310.03302","citing_paper":"/paper/2607.10569"},"observation_digest":"sha256:d3dcc546e481e01a1aafd80fd9b0ad5ccb7086f540fa64e7dee1dbae43eaaf68","observation_id":"a4cb8144-1aa6-4c15-8402-fc415491b0be","resolution":{"observed_at":"2026-07-14T10:46:01.272433Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.03302","last_updated":"2024-04-14T21:02:16Z","snapshot_observed_at":"2026-08-17T14:56:37.479640Z","submitted_at":"2023-10-05T04:06:12Z","title":"MLAgentBench: Evaluating Language Agents on Machine Learning Experimentation","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.03302","snapshot_observed_at":"2026-08-01T23:04:49.393664Z","title":"MLAgentBench: Eval- uating language agents on machine learning experimentation,","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.15541","last_updated":"2026-07-17T01:23:23Z","snapshot_observed_at":"2026-08-16T15:39:09.383753Z","submitted_at":"2026-07-17T01:23:23Z","title":"StarCodex: Dynamic Coding Harness for Starlink Measurement Analysis and Experiment Automation","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-08-01T23:04:49.393664Z"},"links":{"cited_paper":"/paper/2310.03302","citing_paper":"/paper/2607.15541"},"observation_digest":"sha256:6b857693c54733458f206fb7b3a0e5014147f8ef6d60d5b0ce60ae584ff5f9d4","observation_id":"8915d0f7-10cf-4049-800f-45908588bdda","resolution":{"observed_at":"2026-08-01T23:04:49.393664Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.03302","last_updated":"2024-04-14T21:02:16Z","snapshot_observed_at":"2026-08-17T14:56:37.479640Z","submitted_at":"2023-10-05T04:06:12Z","title":"MLAgentBench: Evaluating Language Agents on Machine Learning Experimentation","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.03302","snapshot_observed_at":"2026-08-01T03:35:43.019419Z","title":"MLAgentBench: Evaluating language agents on machine learning experimentation.arXiv preprint arXiv:2310.03302, 2023","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.23123","last_updated":"2026-07-25T09:55:31Z","snapshot_observed_at":"2026-08-13T12:34:51.875986Z","submitted_at":"2026-07-25T09:55:31Z","title":"SQBench: A Benchmark for Evaluating Task Delivery by Language-Model Agents in Production-Oriented Workflows","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-08-01T03:35:43.019419Z"},"links":{"cited_paper":"/paper/2310.03302","citing_paper":"/paper/2607.23123"},"observation_digest":"sha256:87d7bf529048b9755d2c5dcec43fc8774a273a3ae2c9e00d45b60eae1aa47883","observation_id":"75e97664-e545-4253-a3ab-ac63e16441ba","resolution":{"observed_at":"2026-08-01T03:35:43.019419Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.03302","last_updated":"2024-04-14T21:02:16Z","snapshot_observed_at":"2026-08-17T14:56:37.479640Z","submitted_at":"2023-10-05T04:06:12Z","title":"MLAgentBench: Evaluating Language Agents on Machine Learning Experimentation","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.03302","snapshot_observed_at":"2026-07-31T01:39:47.207858Z","title":"Huang, J","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.25090","last_updated":"2026-07-27T21:30:39Z","snapshot_observed_at":"2026-08-13T16:20:37.136961Z","submitted_at":"2026-07-27T21:30:39Z","title":"Matryoshka Agent: Unfolding Sub-Agents for Long-Horizon Machine Learning Engineering","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-07-31T01:39:47.207858Z"},"links":{"cited_paper":"/paper/2310.03302","citing_paper":"/paper/2607.25090"},"observation_digest":"sha256:76978ac2f05172d62c63afaba4dac92f27dd515610f3262ae0b9b3b740d91f93","observation_id":"668008f8-d1d1-4c27-9fb9-3c117585f4d5","resolution":{"observed_at":"2026-07-31T01:39:47.207858Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.03302","last_updated":"2024-04-14T21:02:16Z","snapshot_observed_at":"2026-08-17T14:56:37.479640Z","submitted_at":"2023-10-05T04:06:12Z","title":"MLAgentBench: Evaluating Language Agents on Machine Learning Experimentation","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.03302","snapshot_observed_at":"2026-08-03T11:00:13.304321Z","title":null,"venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2607.29241","last_updated":"2026-07-31T10:15:38Z","snapshot_observed_at":"2026-08-09T09:39:13.620027Z","submitted_at":"2026-07-31T10:15:38Z","title":"RecHarness: A Bandit-Routed Agentic Harness for Self-Evolving Recommender Systems","version":1},"reference_index":63,"source":"arxiv_source","source_observed_at":"2026-08-03T11:00:13.304321Z"},"links":{"cited_paper":"/paper/2310.03302","citing_paper":"/paper/2607.29241"},"observation_digest":"sha256:ce874461aa72c3787826d6ae3fc42b8683bdb676c330bbd6242726671676cbe0","observation_id":"6e976994-74ce-40cc-be05-cb29731d9f6d","resolution":{"observed_at":"2026-08-03T11:00:13.304321Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.03302","last_updated":"2024-04-14T21:02:16Z","snapshot_observed_at":"2026-08-17T14:56:37.479640Z","submitted_at":"2023-10-05T04:06:12Z","title":"MLAgentBench: Evaluating Language Agents on Machine Learning Experimentation","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.03302","snapshot_observed_at":"2026-08-04T00:51:32.809698Z","title":"Benchmarking large language models as ai research agents","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2608.00316","last_updated":"2026-08-11T19:21:07Z","snapshot_observed_at":"2026-08-15T23:09:36.197433Z","submitted_at":"2026-07-31T22:01:18Z","title":"Agentic Bayesian Optimization through Surrogate-Augmented Autoresearch","version":1},"reference_index":35,"source":"arxiv_source","source_observed_at":"2026-08-04T00:51:32.809698Z"},"links":{"cited_paper":"/paper/2310.03302","citing_paper":"/paper/2608.00316"},"observation_digest":"sha256:9704c37cf093ca4742ac54778b6d5e76c85095935fa9b3842bc21c93ae9d406e","observation_id":"bfabdb40-308d-4ade-9774-d158da25aaf2","resolution":{"observed_at":"2026-08-04T00:51:32.809698Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.03302","last_updated":"2024-04-14T21:02:16Z","snapshot_observed_at":"2026-08-17T14:56:37.479640Z","submitted_at":"2023-10-05T04:06:12Z","title":"MLAgentBench: Evaluating Language Agents on Machine Learning Experimentation","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.03302","snapshot_observed_at":"2026-08-15T15:02:48.155720Z","title":"Mlagentbench: Evaluating language agents on machine learning experimentation","venue":null,"work_id":null,"year":2023},"citing_paper":{"arxiv_id":"2608.02775","last_updated":"2026-08-03T18:18:39Z","snapshot_observed_at":"2026-08-17T15:15:23.757266Z","submitted_at":"2026-08-03T18:18:39Z","title":"Towards a new paradigm of scientific discovery with socialized artificial intelligence","version":1},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-08-15T15:02:48.155720Z"},"links":{"cited_paper":"/paper/2310.03302","citing_paper":"/paper/2608.02775"},"observation_digest":"sha256:0a2cfaea43d1c07e805e4f93a9a58ff19aa4d887a8f1d4c8346ed916b62fb20a","observation_id":"b1873b8c-4f30-4464-b184-fb473de7d847","resolution":{"observed_at":"2026-08-15T15:02:48.155720Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.03302","last_updated":"2024-04-14T21:02:16Z","snapshot_observed_at":"2026-08-17T14:56:37.479640Z","submitted_at":"2023-10-05T04:06:12Z","title":"MLAgentBench: Evaluating Language Agents on Machine Learning Experimentation","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.03302","snapshot_observed_at":"2026-08-05T15:25:40.147268Z","title":"doi:10.48550/arXiv.2310.03302 , abstract =","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.03644","last_updated":"2026-08-04T13:29:00Z","snapshot_observed_at":"2026-08-17T17:04:45.145124Z","submitted_at":"2026-08-04T13:29:00Z","title":"Is Inter-Seed Cross-Play Enough? Evaluating the Robustness of Zero-Shot Coordination Algorithms to Implementation Details","version":1},"reference_index":186,"source":"arxiv_source","source_observed_at":"2026-08-05T15:25:40.147268Z"},"links":{"cited_paper":"/paper/2310.03302","citing_paper":"/paper/2608.03644"},"observation_digest":"sha256:c2e2e158d30c45adc9c5a358c74e7c56fd18179db95e1fcb50946079dd0b2b8c","observation_id":"f6a76272-45cc-4970-8ec7-f3dfe9430d26","resolution":{"observed_at":"2026-08-05T15:25:40.147268Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.03302","last_updated":"2024-04-14T21:02:16Z","snapshot_observed_at":"2026-08-17T14:56:37.479640Z","submitted_at":"2023-10-05T04:06:12Z","title":"MLAgentBench: Evaluating Language Agents on Machine Learning Experimentation","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.03302","snapshot_observed_at":"2026-08-11T23:44:24.114790Z","title":"Mlagentbench: Evaluating language agents on machine learning experimentation.arXiv preprint arXiv:2310.03302,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.09096","last_updated":"2026-08-11T02:53:55Z","snapshot_observed_at":"2026-08-16T17:03:39.454138Z","submitted_at":"2026-08-10T03:49:28Z","title":"Evo-Bench: Can Language Models Improve Agent Harness?","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-11T23:44:24.114790Z"},"links":{"cited_paper":"/paper/2310.03302","citing_paper":"/paper/2608.09096"},"observation_digest":"sha256:cd33cf96ebea8aad7b58d17fed8275673725c9521beffe3e027e3ea52e520107","observation_id":"b1f909b1-6832-467f-b923-e52a84a786e0","resolution":{"observed_at":"2026-08-11T23:44:24.114790Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.03302","last_updated":"2024-04-14T21:02:16Z","snapshot_observed_at":"2026-08-17T14:56:37.479640Z","submitted_at":"2023-10-05T04:06:12Z","title":"MLAgentBench: Evaluating Language Agents on Machine Learning Experimentation","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.03302","snapshot_observed_at":"2026-08-14T04:22:26.653121Z","title":"Mlagentbench: Evaluating language agents on machine learning experimentation.arXiv preprint arXiv:2310.03302,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.09096","last_updated":"2026-08-11T02:53:55Z","snapshot_observed_at":"2026-08-16T17:03:39.454138Z","submitted_at":"2026-08-10T03:49:28Z","title":"Evo-Bench: Can Language Models Improve Agent Harness?","version":2},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-08-14T04:22:26.653121Z"},"links":{"cited_paper":"/paper/2310.03302","citing_paper":"/paper/2608.09096"},"observation_digest":"sha256:8bfc9d56eda121e0e6c24fc6c4562dcbbf942ec8f2f30997875a08a2b31c28e1","observation_id":"5f998b68-8bfa-4d4e-b156-2a4bcfd2b75a","resolution":{"observed_at":"2026-08-14T04:22:26.653121Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.03302","last_updated":"2024-04-14T21:02:16Z","snapshot_observed_at":"2026-08-17T14:56:37.479640Z","submitted_at":"2023-10-05T04:06:12Z","title":"MLAgentBench: Evaluating Language Agents on Machine Learning Experimentation","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.03302","snapshot_observed_at":"2026-08-11T15:34:09.138177Z","title":"Mlagentbench: Evaluating language agents on machine learning experimentation.arXiv preprint arXiv:2310.03302,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.09537","last_updated":"2026-08-10T12:35:59Z","snapshot_observed_at":"2026-08-16T01:18:04.231814Z","submitted_at":"2026-08-10T12:35:59Z","title":"verdi: retrieval is not transfer for continual world model optimization","version":1},"reference_index":2026,"source":"pdf_text","source_observed_at":"2026-08-11T15:34:09.138177Z"},"links":{"cited_paper":"/paper/2310.03302","citing_paper":"/paper/2608.09537"},"observation_digest":"sha256:b9627d0a263ac76c7baa2df6b0a2de85ad9f270f50e52131935791a06600e745","observation_id":"98d398fb-3d0b-4ab0-8dfb-c691452f1b71","resolution":{"observed_at":"2026-08-11T15:34:09.138177Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.03302","last_updated":"2024-04-14T21:02:16Z","snapshot_observed_at":"2026-08-17T14:56:37.479640Z","submitted_at":"2023-10-05T04:06:12Z","title":"MLAgentBench: Evaluating Language Agents on Machine Learning Experimentation","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.03302","snapshot_observed_at":"2026-08-15T14:25:46.440154Z","title":"arXiv preprint arXiv:2310.03302 , year=","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.10366","last_updated":"2026-08-11T01:45:56Z","snapshot_observed_at":"2026-08-16T01:39:46.189101Z","submitted_at":"2026-08-11T01:45:56Z","title":"DSAgentBench: Can Agents Automate End-to-End Data-Science Workflows in Real Computer Environments?","version":1},"reference_index":70,"source":"arxiv_source","source_observed_at":"2026-08-15T14:25:46.440154Z"},"links":{"cited_paper":"/paper/2310.03302","citing_paper":"/paper/2608.10366"},"observation_digest":"sha256:8328122f32e09af3f6e16edda3ba4c89a176a9153d322e359518463a9e7d6c1f","observation_id":"2a32d19f-9c42-4377-8d46-6418a4e133d2","resolution":{"observed_at":"2026-08-15T14:25:46.440154Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2310.03302","last_updated":"2024-04-14T21:02:16Z","snapshot_observed_at":"2026-08-17T14:56:37.479640Z","submitted_at":"2023-10-05T04:06:12Z","title":"MLAgentBench: Evaluating Language Agents on Machine Learning Experimentation","version":2},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2310.03302","snapshot_observed_at":"2026-08-15T17:44:14.910309Z","title":"2310.03302 , archiveprefix =","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.13060","last_updated":"2026-08-13T10:23:11Z","snapshot_observed_at":"2026-08-16T23:11:33.456734Z","submitted_at":"2026-08-13T10:23:11Z","title":"VALG: An Agentic System for ML Theory Research","version":1},"reference_index":15,"source":"arxiv_source","source_observed_at":"2026-08-15T17:44:14.910309Z"},"links":{"cited_paper":"/paper/2310.03302","citing_paper":"/paper/2608.13060"},"observation_digest":"sha256:f846830b99d374bd16f20e29f1ae7511632c97255a2a497cfbc31730852f7a1c","observation_id":"4f0fc178-9d5c-4096-9224-1b83e2470f95","resolution":{"observed_at":"2026-08-15T17:44:14.910309Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2310.03302/citation-record","integrity":"/paper/2310.03302/integrity","json":"/paper/2310.03302/citation-record.json","paper":"/paper/2310.03302"},"outbound":[],"paper":{"arxiv_id":"2310.03302","last_updated":"2024-04-14T21:02:16Z","latest_version":2,"primary_category":"cs.LG","snapshot_observed_at":"2026-08-17T14:56:37.479640Z","submitted_at":"2023-10-05T04:06:12Z","title":"MLAgentBench: Evaluating Language Agents on Machine Learning Experimentation"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-17T06:30:58.91139+00:00","source":"crossref"},{"observed_at":"2026-08-17T06:30:54.323127+00:00","source":"retraction_watch"}],"thesis":"As of 17 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 76 inbound Pith citation observations for arXiv:2310.03302."}