{"as_of":"2026-08-10T23:43:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:2370d1b96bb27a28b5f9e96808a3e3aa1cc8e2dc6a4bdd071523b8828d672c75","coverage":[{"denominator":0,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":0,"source":"paper_references, paper_reference_links","source_observed_at":null,"state":"measured"},{"denominator":36,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":36,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-10T06:31:04.303077+00:00","state":"measured"},{"denominator":36,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":36,"source":"paper_references, paper_reference_links","source_observed_at":"2026-08-07T15:42:43.090307Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":1,"source":"arxiv_reference","source_observed_at":"2026-08-05T02:28:24.338817Z","state":"measured"}],"external_citation_measurements":[{"count":4,"observed_at":"2026-08-05T02:28:24.338817Z","source":"arxiv_reference"}],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2502.14499","last_updated":"2025-02-20T12:28:23Z","snapshot_observed_at":"2026-08-08T19:15:34.002979Z","submitted_at":"2025-02-20T12:28:23Z","title":"MLGym: A New Framework and Benchmark for Advancing AI Research Agents","version":1},"cited_work":{"arxiv_id":"2502.14499","doi":"10.48550/arxiv.2502.14499","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.14499","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mlgym: A new framework and benchmark for advancing ai research agents","venue":"ArXiv.org","work_id":"0b9c0950-4be9-4a42-a77f-55c8e4a80b1e","year":2025},"citing_paper":{"arxiv_id":"2504.19678","last_updated":"2026-03-06T19:01:27Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-04-28T11:08:22Z","title":"From LLM Reasoning to Autonomous AI Agents: A Comprehensive Review","version":2},"reference_index":122,"source":"pdf_text","source_observed_at":"2026-05-15T02:57:37.873567Z"},"links":{"cited_paper":"/paper/2502.14499","citing_paper":"/paper/2504.19678"},"observation_digest":"sha256:811e10e2858081fb83a27201849dde60c247ae40ee39b3b2f17f65b862df5431","observation_id":"24af59e5-79f5-48b6-a9b5-fdcb8693094c","resolution":{"observed_at":"2026-05-15T02:57:38.473587Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.14499","last_updated":"2025-02-20T12:28:23Z","snapshot_observed_at":"2026-08-08T19:15:34.002979Z","submitted_at":"2025-02-20T12:28:23Z","title":"MLGym: A New Framework and Benchmark for Advancing AI Research Agents","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.14499","snapshot_observed_at":"2026-08-07T15:42:43.090307Z","title":"Foerster, Yoram Bachrach, William Yang Wang, and Roberta Raileanu","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.14148","last_updated":"2025-05-20T09:55:31Z","snapshot_observed_at":"2026-08-09T08:28:24.979628Z","submitted_at":"2025-05-20T09:55:31Z","title":"MM-Agent: LLM as Agents for Real-world Mathematical Modeling Problem","version":1},"reference_index":49,"source":"pdf_text","source_observed_at":"2026-08-07T15:42:43.090307Z"},"links":{"cited_paper":"/paper/2502.14499","citing_paper":"/paper/2505.14148"},"observation_digest":"sha256:1597f91ed01e2cc4c613ad99ba5e48838d1c7a3c34ac22d727361a3c5af92563","observation_id":"f76081b5-bdb8-4bf5-9fc4-b728fddc0f48","resolution":{"observed_at":"2026-08-07T15:42:43.090307Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.14499","last_updated":"2025-02-20T12:28:23Z","snapshot_observed_at":"2026-08-08T19:15:34.002979Z","submitted_at":"2025-02-20T12:28:23Z","title":"MLGym: A New Framework and Benchmark for Advancing AI Research Agents","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.14499","snapshot_observed_at":"2026-08-07T12:18:53.333828Z","title":"Mlgym: A new framework and benchmark for advancing ai research agents","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2505.24876","last_updated":"2026-05-24T03:59:43Z","snapshot_observed_at":"2026-08-07T12:10:13.725980Z","submitted_at":"2025-05-30T17:59:53Z","title":"Agent-X: Evaluating Deep Multimodal Reasoning in Vision-Centric Agentic Tasks","version":2},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-08-07T12:18:53.333828Z"},"links":{"cited_paper":"/paper/2502.14499","citing_paper":"/paper/2505.24876"},"observation_digest":"sha256:cf91aa1d3eda40f5553d05115afd4cf3e9b3c436b49dc1edefb3a776ce2a8c61","observation_id":"52d62860-0f53-413c-8dc2-a3af42c12106","resolution":{"observed_at":"2026-08-07T12:18:53.333828Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.14499","last_updated":"2025-02-20T12:28:23Z","snapshot_observed_at":"2026-08-08T19:15:34.002979Z","submitted_at":"2025-02-20T12:28:23Z","title":"MLGym: A New Framework and Benchmark for Advancing AI Research Agents","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.14499","snapshot_observed_at":"2026-08-07T10:51:57.997567Z","title":"Mlgym: A new framework and benchmark for advancing ai research agents.arXiv preprint arXiv:2502.14499,","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2506.04098","last_updated":"2025-06-10T13:14:14Z","snapshot_observed_at":"2026-08-09T13:12:42.388209Z","submitted_at":"2025-06-04T15:55:27Z","title":"TextAtari: 100K Frames Game Playing with Language Agents","version":2},"reference_index":57,"source":"pdf_text","source_observed_at":"2026-08-07T10:51:57.997567Z"},"links":{"cited_paper":"/paper/2502.14499","citing_paper":"/paper/2506.04098"},"observation_digest":"sha256:b291b2c8885e14203271df591279a72afedbcba055511e971a19f5c624628add","observation_id":"13645ed6-a5b8-4d94-942a-6771b241bbe6","resolution":{"observed_at":"2026-08-07T10:51:57.997567Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.14499","last_updated":"2025-02-20T12:28:23Z","snapshot_observed_at":"2026-08-08T19:15:34.002979Z","submitted_at":"2025-02-20T12:28:23Z","title":"MLGym: A New Framework and Benchmark for Advancing AI Research Agents","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.14499","snapshot_observed_at":"2026-08-07T10:28:47.225949Z","title":"Mlgym: A new framework and benchmark for advancing ai research agents","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.05213","last_updated":"2025-06-05T16:27:49Z","snapshot_observed_at":"2026-08-08T16:04:48.106610Z","submitted_at":"2025-06-05T16:27:49Z","title":"LLM-First Search: Self-Guided Exploration of the Solution Space","version":1},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-08-07T10:28:47.225949Z"},"links":{"cited_paper":"/paper/2502.14499","citing_paper":"/paper/2506.05213"},"observation_digest":"sha256:76c82a06f9b617427b36babb8558ef56a36422bbf3ef98a28386261acd5d9bd5","observation_id":"463cd5b2-c0d1-4b07-9eca-843fb111b21c","resolution":{"observed_at":"2026-08-07T10:28:47.225949Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.14499","last_updated":"2025-02-20T12:28:23Z","snapshot_observed_at":"2026-08-08T19:15:34.002979Z","submitted_at":"2025-02-20T12:28:23Z","title":"MLGym: A New Framework and Benchmark for Advancing AI Research Agents","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.14499","snapshot_observed_at":"2026-08-07T06:00:19.472486Z","title":"Mlgym: A new framework JOURNAL OF LATEX CLASS FILES, VOL. 14, NO. 8, AUGUST 2015 17 and benchmark for advancing ai research agents,","venue":null,"work_id":null,"year":2015},"citing_paper":{"arxiv_id":"2506.11102","last_updated":"2025-06-06T17:52:18Z","snapshot_observed_at":"2026-08-07T05:54:56.593167Z","submitted_at":"2025-06-06T17:52:18Z","title":"Evolutionary Perspectives on the Evaluation of LLM-Based AI Agents: A Comprehensive Survey","version":1},"reference_index":99,"source":"pdf_text","source_observed_at":"2026-08-07T06:00:19.472486Z"},"links":{"cited_paper":"/paper/2502.14499","citing_paper":"/paper/2506.11102"},"observation_digest":"sha256:e28f09f863be8a56044790cc07ba81c82419c69a6fc4582947d874d75a2d0d18","observation_id":"79bccf71-c36d-4f0c-ba6c-fac4d942ced4","resolution":{"observed_at":"2026-08-07T06:00:19.472486Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.14499","last_updated":"2025-02-20T12:28:23Z","snapshot_observed_at":"2026-08-08T19:15:34.002979Z","submitted_at":"2025-02-20T12:28:23Z","title":"MLGym: A New Framework and Benchmark for Advancing AI Research Agents","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.14499","snapshot_observed_at":"2026-08-06T23:35:49.121636Z","title":"Mlgym: A new framework and benchmark for advancing ai research agents.arXiv preprint arXiv:2502.14499, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.17514","last_updated":"2025-06-20T23:37:17Z","snapshot_observed_at":"2026-08-09T01:40:05.683901Z","submitted_at":"2025-06-20T23:37:17Z","title":"Kaleidoscopic Teaming in Multi Agent Simulations","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-08-06T23:35:49.121636Z"},"links":{"cited_paper":"/paper/2502.14499","citing_paper":"/paper/2506.17514"},"observation_digest":"sha256:21fb52207d69f40ac29a394b60b0e29e2dacce9ec3091329fb5a42e1592bd21e","observation_id":"8a9e3bef-3d00-48e7-bb34-5fb578784dcc","resolution":{"observed_at":"2026-08-06T23:35:49.121636Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.14499","last_updated":"2025-02-20T12:28:23Z","snapshot_observed_at":"2026-08-08T19:15:34.002979Z","submitted_at":"2025-02-20T12:28:23Z","title":"MLGym: A New Framework and Benchmark for Advancing AI Research Agents","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.14499","snapshot_observed_at":"2026-08-06T23:26:56.783174Z","title":"Mlgym: A new framework and benchmark for advancing ai research agents","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2506.18096","last_updated":"2025-09-03T15:32:23Z","snapshot_observed_at":"2026-08-06T23:20:46.744749Z","submitted_at":"2025-06-22T16:52:48Z","title":"Deep Research Agents: A Systematic Examination And Roadmap","version":2},"reference_index":74,"source":"pdf_text","source_observed_at":"2026-08-06T23:26:56.783174Z"},"links":{"cited_paper":"/paper/2502.14499","citing_paper":"/paper/2506.18096"},"observation_digest":"sha256:ef38404aa7ffbce890c98e35e2b1a224b557fad4facd51989914d10a5db8ab91","observation_id":"0860c8ab-af57-4a9f-bc7e-58eb1e415f44","resolution":{"observed_at":"2026-08-06T23:26:56.783174Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.14499","last_updated":"2025-02-20T12:28:23Z","snapshot_observed_at":"2026-08-08T19:15:34.002979Z","submitted_at":"2025-02-20T12:28:23Z","title":"MLGym: A New Framework and Benchmark for Advancing AI Research Agents","version":1},"cited_work":{"arxiv_id":"2502.14499","doi":"10.48550/arxiv.2502.14499","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.14499","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mlgym: A new framework and benchmark for advancing ai research agents","venue":"ArXiv.org","work_id":"0b9c0950-4be9-4a42-a77f-55c8e4a80b1e","year":2025},"citing_paper":{"arxiv_id":"2508.07407","last_updated":"2025-08-31T14:55:05Z","snapshot_observed_at":"2026-08-10T13:27:07.744131Z","submitted_at":"2025-08-10T16:07:32Z","title":"A Comprehensive Survey of Self-Evolving AI Agents: A New Paradigm Bridging Foundation Models and Lifelong Agentic Systems","version":2},"reference_index":62,"source":"pdf_text","source_observed_at":"2026-05-15T23:21:42.029285Z"},"links":{"cited_paper":"/paper/2502.14499","citing_paper":"/paper/2508.07407"},"observation_digest":"sha256:8e875075e74b5013f2a477b7e2e10ea2ecf644092c9af1b8aa79fab0003ed29b","observation_id":"9824b926-79df-498e-ab20-0cee30bbb560","resolution":{"observed_at":"2026-05-15T23:21:42.565484Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.14499","last_updated":"2025-02-20T12:28:23Z","snapshot_observed_at":"2026-08-08T19:15:34.002979Z","submitted_at":"2025-02-20T12:28:23Z","title":"MLGym: A New Framework and Benchmark for Advancing AI Research Agents","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.14499","snapshot_observed_at":"2026-08-05T12:11:34.231652Z","title":"Mlgym: A new framework and benchmark for advancing ai research agents","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2509.04499","last_updated":"2025-09-02T00:32:38Z","snapshot_observed_at":"2026-08-09T20:35:53.900959Z","submitted_at":"2025-09-02T00:32:38Z","title":"DeepTRACE: Auditing Deep Research AI Systems for Tracking Reliability Across Citations and Evidence","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-08-05T12:11:34.231652Z"},"links":{"cited_paper":"/paper/2502.14499","citing_paper":"/paper/2509.04499"},"observation_digest":"sha256:9dc10655fdb9295bb1bddfa69e4bc2e43bb52192decf85285fec0b45db01eb41","observation_id":"9593ae03-a769-400c-bd17-8ebad3f56a97","resolution":{"observed_at":"2026-08-05T12:11:34.231652Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.14499","last_updated":"2025-02-20T12:28:23Z","snapshot_observed_at":"2026-08-08T19:15:34.002979Z","submitted_at":"2025-02-20T12:28:23Z","title":"MLGym: A New Framework and Benchmark for Advancing AI Research Agents","version":1},"cited_work":{"arxiv_id":"2502.14499","doi":"10.48550/arxiv.2502.14499","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.14499","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mlgym: A new framework and benchmark for advancing ai research agents","venue":"ArXiv.org","work_id":"0b9c0950-4be9-4a42-a77f-55c8e4a80b1e","year":2025},"citing_paper":{"arxiv_id":"2604.06111","last_updated":"2026-04-10T03:07:44Z","snapshot_observed_at":"2026-08-02T21:04:18.191344Z","submitted_at":"2026-04-07T17:21:28Z","title":"AgentCE-Bench: Agent Configurable Evaluation with Scalable Horizons and Controllable Difficulty under Lightweight Environments","version":2},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-05-10T19:07:46.077831Z"},"links":{"cited_paper":"/paper/2502.14499","citing_paper":"/paper/2604.06111"},"observation_digest":"sha256:bd8914772ac1ef0422cb9dbc376162a79826bda5e32a0e1d4d654ba1ac326d6d","observation_id":"e9e7f922-74b4-4365-a969-2385d37e313f","resolution":{"observed_at":"2026-05-10T23:30:49.503880Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.14499","last_updated":"2025-02-20T12:28:23Z","snapshot_observed_at":"2026-08-08T19:15:34.002979Z","submitted_at":"2025-02-20T12:28:23Z","title":"MLGym: A New Framework and Benchmark for Advancing AI Research Agents","version":1},"cited_work":{"arxiv_id":"2502.14499","doi":"10.48550/arxiv.2502.14499","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.14499","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mlgym: A new framework and benchmark for advancing ai research agents","venue":"ArXiv.org","work_id":"0b9c0950-4be9-4a42-a77f-55c8e4a80b1e","year":2025},"citing_paper":{"arxiv_id":"2604.06566","last_updated":"2026-04-08T01:34:11Z","snapshot_observed_at":"2026-08-02T20:24:54.387603Z","submitted_at":"2026-04-08T01:34:11Z","title":"AI-Driven Research for Databases","version":1},"reference_index":58,"source":"pdf_text","source_observed_at":"2026-05-10T17:52:17.486258Z"},"links":{"cited_paper":"/paper/2502.14499","citing_paper":"/paper/2604.06566"},"observation_digest":"sha256:9e491b3dd5582c7561004b8ba30939809c4049aea4d96fdff9151e88fb902013","observation_id":"f16f896e-a368-43f4-8557-60b61edd8e3f","resolution":{"observed_at":"2026-05-11T05:55:59.963254Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.14499","last_updated":"2025-02-20T12:28:23Z","snapshot_observed_at":"2026-08-08T19:15:34.002979Z","submitted_at":"2025-02-20T12:28:23Z","title":"MLGym: A New Framework and Benchmark for Advancing AI Research Agents","version":1},"cited_work":{"arxiv_id":"2502.14499","doi":"10.48550/arxiv.2502.14499","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.14499","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mlgym: A new framework and benchmark for advancing ai research agents","venue":"ArXiv.org","work_id":"0b9c0950-4be9-4a42-a77f-55c8e4a80b1e","year":2025},"citing_paper":{"arxiv_id":"2604.14116","last_updated":"2026-04-22T07:33:13Z","snapshot_observed_at":"2026-07-06T23:02:00.082783Z","submitted_at":"2026-04-15T17:38:06Z","title":"TREX: Automating LLM Fine-tuning via Agent-Driven Tree-based Exploration","version":2},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-05-10T12:34:29.808503Z"},"links":{"cited_paper":"/paper/2502.14499","citing_paper":"/paper/2604.14116"},"observation_digest":"sha256:a1ebd04c1a58cc94064658f53661bf123ac56c0a72a25f31045ad2dcf3d46ead","observation_id":"1c404655-6b82-45f4-bfab-9a9a3c85b18f","resolution":{"observed_at":"2026-05-11T11:46:35.120334Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.14499","last_updated":"2025-02-20T12:28:23Z","snapshot_observed_at":"2026-08-08T19:15:34.002979Z","submitted_at":"2025-02-20T12:28:23Z","title":"MLGym: A New Framework and Benchmark for Advancing AI Research Agents","version":1},"cited_work":{"arxiv_id":"2502.14499","doi":"10.48550/arxiv.2502.14499","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.14499","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mlgym: A new framework and benchmark for advancing ai research agents","venue":"ArXiv.org","work_id":"0b9c0950-4be9-4a42-a77f-55c8e4a80b1e","year":2025},"citing_paper":{"arxiv_id":"2605.01250","last_updated":"2026-05-02T05:09:17Z","snapshot_observed_at":"2026-07-06T23:14:29.125042Z","submitted_at":"2026-05-02T05:09:17Z","title":"EO-Gym: A Multimodal, Interactive Environment for Earth Observation Agents","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-05-09T15:03:09.166127Z"},"links":{"cited_paper":"/paper/2502.14499","citing_paper":"/paper/2605.01250"},"observation_digest":"sha256:6f1e849e61dbbf35b2dcedd5dcd6d4b9f4153c9ee5e965c9fdb9ce0cd42dba26","observation_id":"e22e6dff-ce38-43e4-bfff-2fbdc3bd2b30","resolution":{"observed_at":"2026-05-09T22:13:58.361015Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.14499","last_updated":"2025-02-20T12:28:23Z","snapshot_observed_at":"2026-08-08T19:15:34.002979Z","submitted_at":"2025-02-20T12:28:23Z","title":"MLGym: A New Framework and Benchmark for Advancing AI Research Agents","version":1},"cited_work":{"arxiv_id":"2502.14499","doi":"10.48550/arxiv.2502.14499","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.14499","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mlgym: A new framework and benchmark for advancing ai research agents","venue":"ArXiv.org","work_id":"0b9c0950-4be9-4a42-a77f-55c8e4a80b1e","year":2025},"citing_paper":{"arxiv_id":"2605.07021","last_updated":"2026-05-20T02:19:39Z","snapshot_observed_at":"2026-08-02T16:20:14.264105Z","submitted_at":"2026-05-07T23:05:50Z","title":"Behavior Cue Reasoning: Monitorable Reasoning Improves Efficiency and Safety through Oversight","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-05-11T00:54:25.549158Z"},"links":{"cited_paper":"/paper/2502.14499","citing_paper":"/paper/2605.07021"},"observation_digest":"sha256:df733e085bfbf8a535f7e5336099d821955b577ddab658f3621d53624d5bf568","observation_id":"74cfeb6c-6026-46a0-ad7b-a6d6a568ebf1","resolution":{"observed_at":"2026-05-11T05:00:56.553489Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.14499","last_updated":"2025-02-20T12:28:23Z","snapshot_observed_at":"2026-08-08T19:15:34.002979Z","submitted_at":"2025-02-20T12:28:23Z","title":"MLGym: A New Framework and Benchmark for Advancing AI Research Agents","version":1},"cited_work":{"arxiv_id":"2502.14499","doi":"10.48550/arxiv.2502.14499","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.14499","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mlgym: A new framework and benchmark for advancing ai research agents","venue":"ArXiv.org","work_id":"0b9c0950-4be9-4a42-a77f-55c8e4a80b1e","year":2025},"citing_paper":{"arxiv_id":"2605.07021","last_updated":"2026-05-20T02:19:39Z","snapshot_observed_at":"2026-08-02T16:20:14.264105Z","submitted_at":"2026-05-07T23:05:50Z","title":"Behavior Cue Reasoning: Monitorable Reasoning Improves Efficiency and Safety through Oversight","version":2},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-05-21T08:29:09.122055Z"},"links":{"cited_paper":"/paper/2502.14499","citing_paper":"/paper/2605.07021"},"observation_digest":"sha256:82a0cf1bce15577145e303a5d4587c8d9e49429bb65fff219eec72652d506581","observation_id":"16597c5b-5b3f-4237-8ce9-65cc80a6bd3c","resolution":{"observed_at":"2026-05-21T08:29:52.723765Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.14499","last_updated":"2025-02-20T12:28:23Z","snapshot_observed_at":"2026-08-08T19:15:34.002979Z","submitted_at":"2025-02-20T12:28:23Z","title":"MLGym: A New Framework and Benchmark for Advancing AI Research Agents","version":1},"cited_work":{"arxiv_id":"2502.14499","doi":"10.48550/arxiv.2502.14499","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.14499","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mlgym: A new framework and benchmark for advancing ai research agents","venue":"ArXiv.org","work_id":"0b9c0950-4be9-4a42-a77f-55c8e4a80b1e","year":2025},"citing_paper":{"arxiv_id":"2605.08678","last_updated":"2026-07-06T13:36:13Z","snapshot_observed_at":"2026-07-31T14:41:52.251938Z","submitted_at":"2026-05-09T04:29:46Z","title":"MLS-Bench: A Holistic and Rigorous Assessment of AI Systems on Building Better AI","version":1},"reference_index":67,"source":"pdf_text","source_observed_at":"2026-05-12T01:13:35.990078Z"},"links":{"cited_paper":"/paper/2502.14499","citing_paper":"/paper/2605.08678"},"observation_digest":"sha256:cc3ec823eff1e3f505e8a8405a78def85a748e857ccb99b254ec2d5b4ca20728","observation_id":"3343f866-4dd0-4b75-b324-dea8922abc84","resolution":{"observed_at":"2026-05-12T08:21:25.265962Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.14499","last_updated":"2025-02-20T12:28:23Z","snapshot_observed_at":"2026-08-08T19:15:34.002979Z","submitted_at":"2025-02-20T12:28:23Z","title":"MLGym: A New Framework and Benchmark for Advancing AI Research Agents","version":1},"cited_work":{"arxiv_id":"2502.14499","doi":"10.48550/arxiv.2502.14499","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.14499","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mlgym: A new framework and benchmark for advancing ai research agents","venue":"ArXiv.org","work_id":"0b9c0950-4be9-4a42-a77f-55c8e4a80b1e","year":2025},"citing_paper":{"arxiv_id":"2605.08678","last_updated":"2026-07-06T13:36:13Z","snapshot_observed_at":"2026-07-31T14:41:52.251938Z","submitted_at":"2026-05-09T04:29:46Z","title":"MLS-Bench: A Holistic and Rigorous Assessment of AI Systems on Building Better AI","version":2},"reference_index":69,"source":"pdf_text","source_observed_at":"2026-06-30T23:12:57.154537Z"},"links":{"cited_paper":"/paper/2502.14499","citing_paper":"/paper/2605.08678"},"observation_digest":"sha256:f8a7a7b64b0a4e58add641e5eaaf0fe88584044e3fbd1fc3e39ccf03d10d8427","observation_id":"e6a903ea-d8a9-4c5f-a5a4-2ad2b5968289","resolution":{"observed_at":"2026-07-01T13:25:46.030035Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.14499","last_updated":"2025-02-20T12:28:23Z","snapshot_observed_at":"2026-08-08T19:15:34.002979Z","submitted_at":"2025-02-20T12:28:23Z","title":"MLGym: A New Framework and Benchmark for Advancing AI Research Agents","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.14499","snapshot_observed_at":"2026-07-12T17:14:49.310598Z","title":"Foerster, Yoram Bachrach, William Yang Wang, and Roberta Raileanu","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2605.08678","last_updated":"2026-07-06T13:36:13Z","snapshot_observed_at":"2026-07-31T14:41:52.251938Z","submitted_at":"2026-05-09T04:29:46Z","title":"MLS-Bench: A Holistic and Rigorous Assessment of AI Systems on Building Better AI","version":3},"reference_index":68,"source":"pdf_text","source_observed_at":"2026-07-12T17:14:49.310598Z"},"links":{"cited_paper":"/paper/2502.14499","citing_paper":"/paper/2605.08678"},"observation_digest":"sha256:34166d70a5371216645383295ad9bce61bf7cc7456401893833a855f00255052","observation_id":"6ef9c2b0-10b2-4462-942c-e288a2fa4801","resolution":{"observed_at":"2026-07-12T17:14:49.310598Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.14499","last_updated":"2025-02-20T12:28:23Z","snapshot_observed_at":"2026-08-08T19:15:34.002979Z","submitted_at":"2025-02-20T12:28:23Z","title":"MLGym: A New Framework and Benchmark for Advancing AI Research Agents","version":1},"cited_work":{"arxiv_id":"2502.14499","doi":"10.48550/arxiv.2502.14499","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.14499","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mlgym: A new framework and benchmark for advancing ai research agents","venue":"ArXiv.org","work_id":"0b9c0950-4be9-4a42-a77f-55c8e4a80b1e","year":2025},"citing_paper":{"arxiv_id":"2605.16616","last_updated":"2026-05-15T20:35:32Z","snapshot_observed_at":"2026-07-06T23:27:43.762904Z","submitted_at":"2026-05-15T20:35:32Z","title":"MLReplicate: Benchmarking Autonomous Research Systems for Machine Learning Reproducibility","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-05-20T19:59:40.519962Z"},"links":{"cited_paper":"/paper/2502.14499","citing_paper":"/paper/2605.16616"},"observation_digest":"sha256:ce44c1295b5695b90e30e339eaa620fa9e31f89b572ea373f48b66bd2e52eb40","observation_id":"bb6bf2b1-c166-4a23-b57c-4b04936e025b","resolution":{"observed_at":"2026-05-20T20:03:44.085167Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.14499","last_updated":"2025-02-20T12:28:23Z","snapshot_observed_at":"2026-08-08T19:15:34.002979Z","submitted_at":"2025-02-20T12:28:23Z","title":"MLGym: A New Framework and Benchmark for Advancing AI Research Agents","version":1},"cited_work":{"arxiv_id":"2502.14499","doi":"10.48550/arxiv.2502.14499","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.14499","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mlgym: A new framework and benchmark for advancing ai research agents","venue":"ArXiv.org","work_id":"0b9c0950-4be9-4a42-a77f-55c8e4a80b1e","year":2025},"citing_paper":{"arxiv_id":"2605.18661","last_updated":"2026-07-20T17:24:03Z","snapshot_observed_at":"2026-08-02T13:43:29.187658Z","submitted_at":"2026-05-18T17:08:26Z","title":"AI for Auto-Research: Roadmap & User Guide","version":1},"reference_index":135,"source":"pdf_text","source_observed_at":"2026-05-20T10:30:50.256635Z"},"links":{"cited_paper":"/paper/2502.14499","citing_paper":"/paper/2605.18661"},"observation_digest":"sha256:57210e70ed11d442a1212239e045cf603b920359d992de8fed8304739270d5ef","observation_id":"7f48d165-c0ec-4440-8ddd-e272d23f420d","resolution":{"observed_at":"2026-05-20T10:33:12.639602Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.14499","last_updated":"2025-02-20T12:28:23Z","snapshot_observed_at":"2026-08-08T19:15:34.002979Z","submitted_at":"2025-02-20T12:28:23Z","title":"MLGym: A New Framework and Benchmark for Advancing AI Research Agents","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.14499","snapshot_observed_at":"2026-08-02T13:43:40.325453Z","title":"Nathani, L","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2605.18661","last_updated":"2026-07-20T17:24:03Z","snapshot_observed_at":"2026-08-02T13:43:29.187658Z","submitted_at":"2026-05-18T17:08:26Z","title":"AI for Auto-Research: Roadmap & User Guide","version":2},"reference_index":135,"source":"pdf_text","source_observed_at":"2026-08-02T13:43:40.325453Z"},"links":{"cited_paper":"/paper/2502.14499","citing_paper":"/paper/2605.18661"},"observation_digest":"sha256:c2244dac4555a681a870a8237c19b7458dc207b3c3b60b1004881aabe56aadd9","observation_id":"6dc2c65c-838e-40c2-8d4f-b500548d840c","resolution":{"observed_at":"2026-08-02T13:43:40.325453Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.14499","last_updated":"2025-02-20T12:28:23Z","snapshot_observed_at":"2026-08-08T19:15:34.002979Z","submitted_at":"2025-02-20T12:28:23Z","title":"MLGym: A New Framework and Benchmark for Advancing AI Research Agents","version":1},"cited_work":{"arxiv_id":"2502.14499","doi":"10.48550/arxiv.2502.14499","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.14499","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mlgym: A new framework and benchmark for advancing ai research agents","venue":"ArXiv.org","work_id":"0b9c0950-4be9-4a42-a77f-55c8e4a80b1e","year":2025},"citing_paper":{"arxiv_id":"2606.03544","last_updated":"2026-06-02T12:08:38Z","snapshot_observed_at":"2026-07-06T23:43:46.458792Z","submitted_at":"2026-06-02T12:08:38Z","title":"SAGE: A Quantitative Evaluation of Socialized Evolution in Agent Ecosystems","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-06-28T10:02:25.870386Z"},"links":{"cited_paper":"/paper/2502.14499","citing_paper":"/paper/2606.03544"},"observation_digest":"sha256:f27d63b8493d00719fc8295dbac8b75e73d4fa9eeb5c2377a28da79d9553a0be","observation_id":"e24dbd26-fbef-49fe-82f7-367b0c4149cb","resolution":{"observed_at":"2026-07-02T03:26:29.555018Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.14499","last_updated":"2025-02-20T12:28:23Z","snapshot_observed_at":"2026-08-08T19:15:34.002979Z","submitted_at":"2025-02-20T12:28:23Z","title":"MLGym: A New Framework and Benchmark for Advancing AI Research Agents","version":1},"cited_work":{"arxiv_id":"2502.14499","doi":"10.48550/arxiv.2502.14499","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.14499","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mlgym: A new framework and benchmark for advancing ai research agents","venue":"ArXiv.org","work_id":"0b9c0950-4be9-4a42-a77f-55c8e4a80b1e","year":2025},"citing_paper":{"arxiv_id":"2606.07591","last_updated":"2026-07-03T01:40:25Z","snapshot_observed_at":"2026-08-02T19:22:43.861659Z","submitted_at":"2026-05-28T16:27:40Z","title":"ResearchClawBench: A Benchmark for End-to-End Autonomous Scientific Research","version":3},"reference_index":8,"source":"arxiv_source","source_observed_at":"2026-06-29T08:24:42.412763Z"},"links":{"cited_paper":"/paper/2502.14499","citing_paper":"/paper/2606.07591"},"observation_digest":"sha256:3ef8b9a5cee18468f5d2b654098d20179d12e37b1214c529c18148203aceb46f","observation_id":"9cf0516c-52f2-4d55-afa1-a277f036a1c3","resolution":{"observed_at":"2026-06-29T08:33:15.813204Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.14499","last_updated":"2025-02-20T12:28:23Z","snapshot_observed_at":"2026-08-08T19:15:34.002979Z","submitted_at":"2025-02-20T12:28:23Z","title":"MLGym: A New Framework and Benchmark for Advancing AI Research Agents","version":1},"cited_work":{"arxiv_id":"2502.14499","doi":"10.48550/arxiv.2502.14499","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.14499","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mlgym: A new framework and benchmark for advancing ai research agents","venue":"ArXiv.org","work_id":"0b9c0950-4be9-4a42-a77f-55c8e4a80b1e","year":2025},"citing_paper":{"arxiv_id":"2606.07591","last_updated":"2026-07-03T01:40:25Z","snapshot_observed_at":"2026-08-02T19:22:43.861659Z","submitted_at":"2026-05-28T16:27:40Z","title":"ResearchClawBench: A Benchmark for End-to-End Autonomous Scientific Research","version":4},"reference_index":8,"source":"arxiv_source","source_observed_at":"2026-07-04T00:30:00.449270Z"},"links":{"cited_paper":"/paper/2502.14499","citing_paper":"/paper/2606.07591"},"observation_digest":"sha256:e1ed5979ea88dbb886f60c59d848416a52e65965e36bff0f2bbcf8383e0e25e7","observation_id":"2ddd90b3-696c-4d5e-941f-f6a3714ce425","resolution":{"observed_at":"2026-07-04T00:39:16.507175Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.14499","last_updated":"2025-02-20T12:28:23Z","snapshot_observed_at":"2026-08-08T19:15:34.002979Z","submitted_at":"2025-02-20T12:28:23Z","title":"MLGym: A New Framework and Benchmark for Advancing AI Research Agents","version":1},"cited_work":{"arxiv_id":"2502.14499","doi":"10.48550/arxiv.2502.14499","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.14499","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mlgym: A new framework and benchmark for advancing ai research agents","venue":"ArXiv.org","work_id":"0b9c0950-4be9-4a42-a77f-55c8e4a80b1e","year":2025},"citing_paper":{"arxiv_id":"2606.09550","last_updated":"2026-06-08T14:29:40Z","snapshot_observed_at":"2026-07-06T23:48:53.420827Z","submitted_at":"2026-06-08T14:29:40Z","title":"InquiTree: Evaluating AI Agents in the Scientific Inquiry Loop with Paper-Derived Research Trees","version":1},"reference_index":3,"source":"arxiv_source","source_observed_at":"2026-06-27T14:06:59.471772Z"},"links":{"cited_paper":"/paper/2502.14499","citing_paper":"/paper/2606.09550"},"observation_digest":"sha256:7f6ee4d976611d117cd35e4a5a960177407161b906f2b9e54c17bdd2aa6d0639","observation_id":"be25b1b2-71ba-4e48-bac1-dffb1720bbf0","resolution":{"observed_at":"2026-07-03T04:07:37.266203Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.14499","last_updated":"2025-02-20T12:28:23Z","snapshot_observed_at":"2026-08-08T19:15:34.002979Z","submitted_at":"2025-02-20T12:28:23Z","title":"MLGym: A New Framework and Benchmark for Advancing AI Research Agents","version":1},"cited_work":{"arxiv_id":"2502.14499","doi":"10.48550/arxiv.2502.14499","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.14499","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mlgym: A new framework and benchmark for advancing ai research agents","venue":"ArXiv.org","work_id":"0b9c0950-4be9-4a42-a77f-55c8e4a80b1e","year":2025},"citing_paper":{"arxiv_id":"2606.12736","last_updated":"2026-06-10T22:55:30Z","snapshot_observed_at":"2026-08-02T11:05:19.132987Z","submitted_at":"2026-06-10T22:55:30Z","title":"Benchmarking AI Agents for Addressing Scientific Challenges Across Scales","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-06-27T09:34:09.347912Z"},"links":{"cited_paper":"/paper/2502.14499","citing_paper":"/paper/2606.12736"},"observation_digest":"sha256:ba9306b01a23b0f4ffb72d6949514a0f9929e09289b5c172a4b3f9eac5119fc1","observation_id":"9362f5de-c93a-4a7d-ae64-b221944a01d1","resolution":{"observed_at":"2026-07-03T11:28:04.333266Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.14499","last_updated":"2025-02-20T12:28:23Z","snapshot_observed_at":"2026-08-08T19:15:34.002979Z","submitted_at":"2025-02-20T12:28:23Z","title":"MLGym: A New Framework and Benchmark for Advancing AI Research Agents","version":1},"cited_work":{"arxiv_id":"2502.14499","doi":"10.48550/arxiv.2502.14499","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.14499","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mlgym: A new framework and benchmark for advancing ai research agents","venue":"ArXiv.org","work_id":"0b9c0950-4be9-4a42-a77f-55c8e4a80b1e","year":2025},"citing_paper":{"arxiv_id":"2606.13148","last_updated":"2026-07-01T17:31:23Z","snapshot_observed_at":"2026-08-06T18:49:52.353872Z","submitted_at":"2026-06-11T10:26:03Z","title":"TerraBench: Can Agents Reason Over Heterogeneous Earth-System Data?","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-06-27T07:17:43.188693Z"},"links":{"cited_paper":"/paper/2502.14499","citing_paper":"/paper/2606.13148"},"observation_digest":"sha256:71d9fa7ed38ce75f74892f701bf015483d424b29d6c465513a12a09a73c5b888","observation_id":"fe5d0459-1a22-417e-a3b0-1d2c71d1afa2","resolution":{"observed_at":"2026-07-03T13:58:22.222379Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.14499","last_updated":"2025-02-20T12:28:23Z","snapshot_observed_at":"2026-08-08T19:15:34.002979Z","submitted_at":"2025-02-20T12:28:23Z","title":"MLGym: A New Framework and Benchmark for Advancing AI Research Agents","version":1},"cited_work":{"arxiv_id":"2502.14499","doi":"10.48550/arxiv.2502.14499","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.14499","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mlgym: A new framework and benchmark for advancing ai research agents","venue":"ArXiv.org","work_id":"0b9c0950-4be9-4a42-a77f-55c8e4a80b1e","year":2025},"citing_paper":{"arxiv_id":"2606.13148","last_updated":"2026-07-01T17:31:23Z","snapshot_observed_at":"2026-08-06T18:49:52.353872Z","submitted_at":"2026-06-11T10:26:03Z","title":"TerraBench: Can Agents Reason Over Heterogeneous Earth-System Data?","version":2},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-07-02T22:29:58.492918Z"},"links":{"cited_paper":"/paper/2502.14499","citing_paper":"/paper/2606.13148"},"observation_digest":"sha256:f8c454e64709e912ea017886020d1254a0dff5e769d4f1f62df93c40fd9b8fd7","observation_id":"21be2fa4-d534-4af3-9fd3-422ffd310c30","resolution":{"observed_at":"2026-07-02T22:37:25.449237Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.14499","last_updated":"2025-02-20T12:28:23Z","snapshot_observed_at":"2026-08-08T19:15:34.002979Z","submitted_at":"2025-02-20T12:28:23Z","title":"MLGym: A New Framework and Benchmark for Advancing AI Research Agents","version":1},"cited_work":{"arxiv_id":"2502.14499","doi":"10.48550/arxiv.2502.14499","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.14499","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mlgym: A new framework and benchmark for advancing ai research agents","venue":"ArXiv.org","work_id":"0b9c0950-4be9-4a42-a77f-55c8e4a80b1e","year":2025},"citing_paper":{"arxiv_id":"2606.17838","last_updated":"2026-06-16T12:06:27Z","snapshot_observed_at":"2026-08-07T00:51:41.569744Z","submitted_at":"2026-06-16T12:06:27Z","title":"Environment-Grounded Automated Prompt Optimization for LLM Game Agents","version":1},"reference_index":36,"source":"arxiv_source","source_observed_at":"2026-06-27T00:30:42.133192Z"},"links":{"cited_paper":"/paper/2502.14499","citing_paper":"/paper/2606.17838"},"observation_digest":"sha256:22f9341b4c41f614c69b9e3f62c58d2f79de5d259b1800eced1b9ba36da63799","observation_id":"0cd66105-af08-4e2f-87c2-a77ade7a068b","resolution":{"observed_at":"2026-06-27T00:40:18.662768Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.14499","last_updated":"2025-02-20T12:28:23Z","snapshot_observed_at":"2026-08-08T19:15:34.002979Z","submitted_at":"2025-02-20T12:28:23Z","title":"MLGym: A New Framework and Benchmark for Advancing AI Research Agents","version":1},"cited_work":{"arxiv_id":"2502.14499","doi":"10.48550/arxiv.2502.14499","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.14499","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mlgym: A new framework and benchmark for advancing ai research agents","venue":"ArXiv.org","work_id":"0b9c0950-4be9-4a42-a77f-55c8e4a80b1e","year":2025},"citing_paper":{"arxiv_id":"2606.22866","last_updated":"2026-06-22T05:24:10Z","snapshot_observed_at":"2026-08-07T17:23:42.927399Z","submitted_at":"2026-06-22T05:24:10Z","title":"Discovering Crystal Structure Prediction Algorithms with an AI Co-Scientist","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-06-26T09:03:00.675456Z"},"links":{"cited_paper":"/paper/2502.14499","citing_paper":"/paper/2606.22866"},"observation_digest":"sha256:c33525b68a57dc4633a5d6089874447cdf72790e03cb7edd021a05f32ae3dc24","observation_id":"954b2b2f-e649-4138-a210-40e2995b1809","resolution":{"observed_at":"2026-07-04T10:19:46.994246Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.14499","last_updated":"2025-02-20T12:28:23Z","snapshot_observed_at":"2026-08-08T19:15:34.002979Z","submitted_at":"2025-02-20T12:28:23Z","title":"MLGym: A New Framework and Benchmark for Advancing AI Research Agents","version":1},"cited_work":{"arxiv_id":"2502.14499","doi":"10.48550/arxiv.2502.14499","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.14499","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mlgym: A new framework and benchmark for advancing ai research agents","venue":"ArXiv.org","work_id":"0b9c0950-4be9-4a42-a77f-55c8e4a80b1e","year":2025},"citing_paper":{"arxiv_id":"2606.24530","last_updated":"2026-07-06T16:56:53Z","snapshot_observed_at":"2026-07-12T12:32:23.590658Z","submitted_at":"2026-06-23T12:58:23Z","title":"NatureBench: Can Coding Agents Match the Published SOTA of Nature-Family Papers?","version":1},"reference_index":64,"source":"arxiv_source","source_observed_at":"2026-06-26T00:13:14.940915Z"},"links":{"cited_paper":"/paper/2502.14499","citing_paper":"/paper/2606.24530"},"observation_digest":"sha256:5f907d5055fd6ad21ae9b95ca41d6b6a368700265551764f0765ca2593e9e4de","observation_id":"82ea1e40-c1c7-43a5-aaf0-47e24585bd20","resolution":{"observed_at":"2026-07-04T16:49:57.820516Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-10T06:31:04.303077+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.14499","last_updated":"2025-02-20T12:28:23Z","snapshot_observed_at":"2026-08-08T19:15:34.002979Z","submitted_at":"2025-02-20T12:28:23Z","title":"MLGym: A New Framework and Benchmark for Advancing AI Research Agents","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.14499","snapshot_observed_at":"2026-07-12T12:32:39.706055Z","title":"MLGym : A new framework and benchmark for advancing AI research agents, 2025","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2606.24530","last_updated":"2026-07-06T16:56:53Z","snapshot_observed_at":"2026-07-12T12:32:23.590658Z","submitted_at":"2026-06-23T12:58:23Z","title":"NatureBench: Can Coding Agents Match the Published SOTA of Nature-Family Papers?","version":2},"reference_index":64,"source":"arxiv_source","source_observed_at":"2026-07-12T12:32:39.706055Z"},"links":{"cited_paper":"/paper/2502.14499","citing_paper":"/paper/2606.24530"},"observation_digest":"sha256:512564ef22ebd7d7f81cbe514f82a2b289b83ce348b59c4b00a2a3a0331122cf","observation_id":"6fab8d45-91dc-469c-8eeb-8e5e0e8c7039","resolution":{"observed_at":"2026-07-12T12:32:39.706055Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.14499","last_updated":"2025-02-20T12:28:23Z","snapshot_observed_at":"2026-08-08T19:15:34.002979Z","submitted_at":"2025-02-20T12:28:23Z","title":"MLGym: A New Framework and Benchmark for Advancing AI Research Agents","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.14499","snapshot_observed_at":"2026-08-01T16:16:31.350666Z","title":"arXiv preprint arXiv:2502.14499 (2025)","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.18064","last_updated":"2026-07-20T15:32:09Z","snapshot_observed_at":"2026-08-07T19:30:10.496735Z","submitted_at":"2026-07-20T15:32:09Z","title":"Autoresearch with Coding Agents: Generalizers and Metric-Maximizers on Quran Recitation Data","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-08-01T16:16:31.350666Z"},"links":{"cited_paper":"/paper/2502.14499","citing_paper":"/paper/2607.18064"},"observation_digest":"sha256:752ce37e67e200cddf9748fb686f0ed885c92288393b6b87287a91567bc1ff10","observation_id":"db995256-b2bf-4c09-bff4-0a297d14f7e0","resolution":{"observed_at":"2026-08-01T16:16:31.350666Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.14499","last_updated":"2025-02-20T12:28:23Z","snapshot_observed_at":"2026-08-08T19:15:34.002979Z","submitted_at":"2025-02-20T12:28:23Z","title":"MLGym: A New Framework and Benchmark for Advancing AI Research Agents","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.14499","snapshot_observed_at":"2026-08-01T12:50:04.327582Z","title":"2025 , url =","venue":null,"work_id":null,"year":2025},"citing_paper":{"arxiv_id":"2607.26587","last_updated":"2026-07-29T08:06:43Z","snapshot_observed_at":"2026-08-07T15:35:26.187389Z","submitted_at":"2026-07-29T08:06:43Z","title":"One Run Is Not an Idea: The Implementation Lottery in Automated Research","version":1},"reference_index":8,"source":"arxiv_source","source_observed_at":"2026-08-01T12:50:04.327582Z"},"links":{"cited_paper":"/paper/2502.14499","citing_paper":"/paper/2607.26587"},"observation_digest":"sha256:681fce804a6bf85341d29b9b957aee0740aa8d99e7bffff4a02c1c03d6e98523","observation_id":"90d01cd4-8504-4389-8829-49cd3e2e4d33","resolution":{"observed_at":"2026-08-01T12:50:04.327582Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}},{"citation":{"cited_paper":{"arxiv_id":"2502.14499","last_updated":"2025-02-20T12:28:23Z","snapshot_observed_at":"2026-08-08T19:15:34.002979Z","submitted_at":"2025-02-20T12:28:23Z","title":"MLGym: A New Framework and Benchmark for Advancing AI Research Agents","version":1},"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":null,"pith_arxiv_id":"2502.14499","snapshot_observed_at":"2026-08-05T15:25:40.067169Z","title":"doi:10.48550/arXiv.2502.14499 , abstract =","venue":null,"work_id":null,"year":null},"citing_paper":{"arxiv_id":"2608.03644","last_updated":"2026-08-04T13:29:00Z","snapshot_observed_at":"2026-08-08T18:58:23.959751Z","submitted_at":"2026-08-04T13:29:00Z","title":"Is Inter-Seed Cross-Play Enough? Evaluating the Robustness of Zero-Shot Coordination Algorithms to Implementation Details","version":1},"reference_index":169,"source":"arxiv_source","source_observed_at":"2026-08-05T15:25:40.067169Z"},"links":{"cited_paper":"/paper/2502.14499","citing_paper":"/paper/2608.03644"},"observation_digest":"sha256:7a46f4879019cdc5668b55eba9a99c32722987bc89630a039d6d8f3a138cae00","observation_id":"49b22e7f-9d05-42f0-907f-9ffcb04fa167","resolution":{"observed_at":"2026-08-05T15:25:40.067169Z","resolver_source":null,"status":"unresolved"},"standing_notice":{"events":[],"reason":"canonical_work_link_unavailable","source_receipts":[],"state":"unavailable"}}],"links":{"evidence":"/evidence","html":"/paper/2502.14499/citation-record","integrity":"/paper/2502.14499/integrity","json":"/paper/2502.14499/citation-record.json","paper":"/paper/2502.14499"},"outbound":[],"paper":{"arxiv_id":"2502.14499","last_updated":"2025-02-20T12:28:23Z","latest_version":1,"primary_category":"cs.CL","snapshot_observed_at":"2026-08-08T19:15:34.002979Z","submitted_at":"2025-02-20T12:28:23Z","title":"MLGym: A New Framework and Benchmark for Advancing AI Research Agents"},"reference_resolution":{"displayed":0,"state_counts":{"malformed_identifier":0,"metadata_mismatch":0,"parse_uncertain":0,"unresolved":0,"verified_exact":0,"verified_fuzzy":0},"total_outbound_references":0},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-10T06:31:04.303077+00:00","source":"crossref"},{"observed_at":"2026-08-10T06:30:57.382061+00:00","source":"retraction_watch"}],"thesis":"As of 10 August 2026, this Paper Citation Record lists 0 of 0 outbound references and 36 inbound Pith citation observations for arXiv:2502.14499."}