{"as_of":"2026-08-22T14:42:00Z","caps":{"database_statements":6,"inbound":100,"outbound":100},"context_digest":"sha256:e3530b3772b9f22216be19174955bfce4984494cdffc175a3a9d0694fb79c06e","coverage":[{"denominator":43,"lane":"reference_resolution","note":"Typed states for the displayed outbound observations.","records_observed":43,"source":"paper_references, paper_reference_links","source_observed_at":"2026-05-20T19:59:40.519962Z","state":"measured"},{"denominator":44,"lane":"standing_notices","note":"One-hop event checks from named stored sources.","records_observed":44,"source":"scholarly_work_events, retraction_status_cache","source_observed_at":"2026-08-22T06:32:14.747728+00:00","state":"measured"},{"denominator":1,"lane":"inbound_itemization","note":"Pith citing papers itemized under the disclosed page cap.","records_observed":1,"source":"paper_references, paper_reference_links","source_observed_at":"2026-07-09T03:36:57.168246Z","state":"measured"},{"denominator":1,"lane":"external_citation_measurements","note":"A source-named dated measurement, never combined with another source.","records_observed":0,"source":"pith","source_observed_at":"2026-07-09T03:45:55.800152Z","state":"measured"}],"external_citation_measurements":[],"inbound":[{"citation":{"cited_paper":{"arxiv_id":"2605.16616","last_updated":"2026-05-15T20:35:32Z","snapshot_observed_at":"2026-08-16T06:46:32.334621Z","submitted_at":"2026-05-15T20:35:32Z","title":"MLReplicate: Benchmarking Autonomous Research Systems for Machine Learning Reproducibility","version":1},"cited_work":{"arxiv_id":"2605.16616","doi":null,"metadata_source":"pith","pith_arxiv_id":"2605.16616","snapshot_observed_at":"2026-07-09T03:45:55.800152Z","title":"MLReplicate: Benchmarking Autonomous Research Systems for Machine Learning Reproducibility","venue":"cs.LG","work_id":"6d1a8776-5bdc-40da-bfa6-cbea2ed82e21","year":2026},"citing_paper":{"arxiv_id":"2607.07663","last_updated":"2026-07-08T17:19:50Z","snapshot_observed_at":"2026-08-12T14:05:29.940431Z","submitted_at":"2026-07-08T17:19:50Z","title":"Recursive Self-Improvement in AI: From Bounded Self-Refinement to Autonomous Research Loops","version":1},"reference_index":190,"source":"pdf_text","source_observed_at":"2026-07-09T03:36:57.168246Z"},"links":{"cited_paper":"/paper/2605.16616","citing_paper":"/paper/2607.07663"},"observation_digest":"sha256:d289a4e3728ab6dff888783fc9ba236d2ca141f4a76fabd6f8cb1fc51174f251","observation_id":"05e10731-bada-4fa4-8566-3aca3ab0fcc2","resolution":{"observed_at":"2026-07-09T03:45:55.801393Z","resolver_source":"local_arxiv","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}}],"links":{"evidence":"/evidence","html":"/paper/2605.16616/citation-record","integrity":"/paper/2605.16616/integrity","json":"/paper/2605.16616/citation-record.json","paper":"/paper/2605.16616"},"outbound":[{"citation":{"cited_paper":{"arxiv_id":"2303.08774","last_updated":"2024-03-04T06:01:33Z","snapshot_observed_at":"2026-08-17T09:58:46.058102Z","submitted_at":"2023-03-15T17:15:04Z","title":"GPT-4 Technical Report","version":6},"cited_work":{"arxiv_id":"2303.08774","doi":"10.1002/tea.20265","metadata_source":"pith","pith_arxiv_id":"2303.08774","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"GPT-4 Technical Report","venue":"cs.CL","work_id":"b928e041-6991-4c08-8c81-0359e4097c7b","year":2023},"citing_paper":{"arxiv_id":"2605.16616","last_updated":"2026-05-15T20:35:32Z","snapshot_observed_at":"2026-08-16T06:46:32.334621Z","submitted_at":"2026-05-15T20:35:32Z","title":"MLReplicate: Benchmarking Autonomous Research Systems for Machine Learning Reproducibility","version":1},"reference_index":1,"source":"pdf_text","source_observed_at":"2026-05-20T19:59:40.519962Z"},"links":{"cited_paper":"/paper/2303.08774","citing_paper":"/paper/2605.16616"},"observation_digest":"sha256:411803842e7b3adcfd9120449ca066a19f5af392a6659432b022d802c5956402","observation_id":"1b166c4b-c536-4ebf-8064-ab6cd53f0cf6","resolution":{"observed_at":"2026-05-20T20:03:44.113477Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"InProceedings of the 2024 Conference on Empirical Methods in Natural Language Processing","venue":null,"work_id":"523ececa-93b5-46f9-b9cc-417415fbf44c","year":2024},"citing_paper":{"arxiv_id":"2605.16616","last_updated":"2026-05-15T20:35:32Z","snapshot_observed_at":"2026-08-16T06:46:32.334621Z","submitted_at":"2026-05-15T20:35:32Z","title":"MLReplicate: Benchmarking Autonomous Research Systems for Machine Learning Reproducibility","version":1},"reference_index":2,"source":"pdf_text","source_observed_at":"2026-05-20T19:59:40.519962Z"},"links":{"citing_paper":"/paper/2605.16616"},"observation_digest":"sha256:25f5ae7a15c6c9617f9fc3ac4dfd8c0a6ad10619a201ea895cf65e5e9fe14477","observation_id":"e31bd759-e308-415c-ad3c-a810caa9e9bf","resolution":{"observed_at":"2026-05-20T20:03:59.403879Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Andres M Bran, Sam Cox, Oliver Schilter, Carlo Baldassari, Andrew D White, and Philippe Schwaller","venue":null,"work_id":"a63809ce-615a-4a73-b95c-cee958aadee2","year":2023},"citing_paper":{"arxiv_id":"2605.16616","last_updated":"2026-05-15T20:35:32Z","snapshot_observed_at":"2026-08-16T06:46:32.334621Z","submitted_at":"2026-05-15T20:35:32Z","title":"MLReplicate: Benchmarking Autonomous Research Systems for Machine Learning Reproducibility","version":1},"reference_index":3,"source":"pdf_text","source_observed_at":"2026-05-20T19:59:40.519962Z"},"links":{"citing_paper":"/paper/2605.16616"},"observation_digest":"sha256:e965c0137b4c5e7ce867aacfb9f4f75dc6ac28aef44ed58e6e13a108668f1ac1","observation_id":"82dd0b20-74ad-4494-859a-e198958f7e08","resolution":{"observed_at":"2026-05-20T20:03:59.396093Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2304.05376","last_updated":"2023-10-02T17:03:01Z","snapshot_observed_at":"2026-08-12T21:57:52.033689Z","submitted_at":"2023-04-11T17:41:13Z","title":"ChemCrow: Augmenting large-language models with chemistry tools","version":5},"cited_work":{"arxiv_id":"2304.05376","doi":"10.48550/arxiv.2304.05376","metadata_source":"pith","pith_arxiv_id":"2304.05376","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"ChemCrow: Augmenting large-language models with chemistry tools","venue":"physics.chem-ph","work_id":"1ceaeb05-7517-4fbd-ba9a-7a63c96b23e6","year":2023},"citing_paper":{"arxiv_id":"2605.16616","last_updated":"2026-05-15T20:35:32Z","snapshot_observed_at":"2026-08-16T06:46:32.334621Z","submitted_at":"2026-05-15T20:35:32Z","title":"MLReplicate: Benchmarking Autonomous Research Systems for Machine Learning Reproducibility","version":1},"reference_index":4,"source":"pdf_text","source_observed_at":"2026-05-20T19:59:40.519962Z"},"links":{"cited_paper":"/paper/2304.05376","citing_paper":"/paper/2605.16616"},"observation_digest":"sha256:ce14bcc7ba18e8d4446197ec684949b5cdf392ac314c6a864c889888074a7e5a","observation_id":"58d4332f-1e00-4c80-b0c8-e5c3f61908f9","resolution":{"observed_at":"2026-05-20T20:03:44.099705Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Jun Shern Chan, Neil Chowdhury, Oliver Jaffe, James Aung, Dane Sherburn, Evan Mays, Giulio Starace, Kevin Liu, Leon Maksin, Tejal Patwardhan, et al","venue":null,"work_id":"f6bf9e53-3979-422d-b0d0-e4f3e136d1f8","year":2020},"citing_paper":{"arxiv_id":"2605.16616","last_updated":"2026-05-15T20:35:32Z","snapshot_observed_at":"2026-08-16T06:46:32.334621Z","submitted_at":"2026-05-15T20:35:32Z","title":"MLReplicate: Benchmarking Autonomous Research Systems for Machine Learning Reproducibility","version":1},"reference_index":5,"source":"pdf_text","source_observed_at":"2026-05-20T19:59:40.519962Z"},"links":{"citing_paper":"/paper/2605.16616"},"observation_digest":"sha256:31f3bd687797929cab72339f1950f5333e1bbebff935f0702de39b48c14eb599","observation_id":"04fe23f3-d80f-4a3b-a8a1-d1102e13cd29","resolution":{"observed_at":"2026-05-20T20:03:59.407414Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.07095","last_updated":"2025-02-26T11:57:30Z","snapshot_observed_at":"2026-08-12T16:49:10.970507Z","submitted_at":"2024-10-09T17:34:27Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","version":6},"cited_work":{"arxiv_id":"2410.07095","doi":"10.48550/arxiv.2410.07095","metadata_source":"pith","pith_arxiv_id":"2410.07095","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","venue":"cs.CL","work_id":"a671e43f-ceab-49e7-adc3-473d802a97ca","year":2024},"citing_paper":{"arxiv_id":"2605.16616","last_updated":"2026-05-15T20:35:32Z","snapshot_observed_at":"2026-08-16T06:46:32.334621Z","submitted_at":"2026-05-15T20:35:32Z","title":"MLReplicate: Benchmarking Autonomous Research Systems for Machine Learning Reproducibility","version":1},"reference_index":6,"source":"pdf_text","source_observed_at":"2026-05-20T19:59:40.519962Z"},"links":{"cited_paper":"/paper/2410.07095","citing_paper":"/paper/2605.16616"},"observation_digest":"sha256:0b2c8ea0282c634aa44da11f2fe53db5cb25f56c24d141e83f5ee64816668f52","observation_id":"92bd12e4-4819-4baa-8d3b-88a8a9a2219f","resolution":{"observed_at":"2026-05-20T20:03:44.137430Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2505.19955","doi":"10.48550/arxiv.2505.19955","metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"arXiv preprint arXiv:2505.19955(2025)","venue":"ArXiv.org","work_id":"5096c958-8775-4f54-ac19-c33b3bca2724","year":2025},"citing_paper":{"arxiv_id":"2605.16616","last_updated":"2026-05-15T20:35:32Z","snapshot_observed_at":"2026-08-16T06:46:32.334621Z","submitted_at":"2026-05-15T20:35:32Z","title":"MLReplicate: Benchmarking Autonomous Research Systems for Machine Learning Reproducibility","version":1},"reference_index":7,"source":"pdf_text","source_observed_at":"2026-05-20T19:59:40.519962Z"},"links":{"citing_paper":"/paper/2605.16616"},"observation_digest":"sha256:40e79bc01203e67310a48e94f89093c2972ea565fe4e149363bc291b02672177","observation_id":"62a52266-d15c-461c-9fe8-8fde1897bea2","resolution":{"observed_at":"2026-05-20T20:03:44.095253Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2410.05080","last_updated":"2025-03-31T14:39:44Z","snapshot_observed_at":"2026-08-16T13:11:53.516904Z","submitted_at":"2024-10-07T14:33:50Z","title":"ScienceAgentBench: Toward Rigorous Assessment of Language Agents for Data-Driven Scientific Discovery","version":3},"cited_work":{"arxiv_id":"2410.05080","doi":"10.48550/arxiv.2410.05080","metadata_source":"pith","pith_arxiv_id":"2410.05080","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Scienceagentbench: Toward rigorous assessment of language agents for data-driven scientific discovery","venue":"cs.CL","work_id":"e1160394-4fe6-4475-9753-e69ad39fb09e","year":2024},"citing_paper":{"arxiv_id":"2605.16616","last_updated":"2026-05-15T20:35:32Z","snapshot_observed_at":"2026-08-16T06:46:32.334621Z","submitted_at":"2026-05-15T20:35:32Z","title":"MLReplicate: Benchmarking Autonomous Research Systems for Machine Learning Reproducibility","version":1},"reference_index":8,"source":"pdf_text","source_observed_at":"2026-05-20T19:59:40.519962Z"},"links":{"cited_paper":"/paper/2410.05080","citing_paper":"/paper/2605.16616"},"observation_digest":"sha256:4f1b998592dfaf8b80e1ce2ac14a3b9cfe7b91172adbd1ae052db75de2788ece","observation_id":"9ae8b600-47c5-42ad-9d29-ecefac5a3aa4","resolution":{"observed_at":"2026-05-20T20:03:44.130932Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-05-20T23:23:21.375775+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-20T23:23:21.375775+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2501.19334","last_updated":"2025-07-11T17:08:45Z","snapshot_observed_at":"2026-08-16T22:21:37.145468Z","submitted_at":"2025-01-31T17:34:53Z","title":"The Value of Prediction in Identifying the Worst-Off","version":3},"cited_work":{"arxiv_id":"2501.19334","doi":"10.48550/arxiv.2501.19334","metadata_source":"arxiv_reference","pith_arxiv_id":"2501.19334","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Alireza Ghafarollahi and Markus J Buehler","venue":"ArXiv.org","work_id":"5227514e-7776-4de8-826e-5ee50f10013c","year":2025},"citing_paper":{"arxiv_id":"2605.16616","last_updated":"2026-05-15T20:35:32Z","snapshot_observed_at":"2026-08-16T06:46:32.334621Z","submitted_at":"2026-05-15T20:35:32Z","title":"MLReplicate: Benchmarking Autonomous Research Systems for Machine Learning Reproducibility","version":1},"reference_index":9,"source":"pdf_text","source_observed_at":"2026-05-20T19:59:40.519962Z"},"links":{"cited_paper":"/paper/2501.19334","citing_paper":"/paper/2605.16616"},"observation_digest":"sha256:970e60c6adb8f1ed7959afb4c3f65684f5baa9f40ef9fbf2d44f1a0f6d2da9c8","observation_id":"59815ca7-b530-4067-b208-b88c3b69b009","resolution":{"observed_at":"2026-05-20T20:03:44.104273Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Josh Givens, Song Liu, and Henry WJ Reeve","venue":null,"work_id":"d9c6791e-995e-478d-a22c-67d34b1aa8be","year":2025},"citing_paper":{"arxiv_id":"2605.16616","last_updated":"2026-05-15T20:35:32Z","snapshot_observed_at":"2026-08-16T06:46:32.334621Z","submitted_at":"2026-05-15T20:35:32Z","title":"MLReplicate: Benchmarking Autonomous Research Systems for Machine Learning Reproducibility","version":1},"reference_index":10,"source":"pdf_text","source_observed_at":"2026-05-20T19:59:40.519962Z"},"links":{"citing_paper":"/paper/2605.16616"},"observation_digest":"sha256:f59a03b6064f47e1b7364d4e788efa460dec37b598de1cfed3e3b74db5adc08e","observation_id":"ca2e8f01-3988-47cc-910c-fb5e7c0cc65e","resolution":{"observed_at":"2026-05-20T20:03:59.400467Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.00557","last_updated":"2025-05-31T13:26:51Z","snapshot_observed_at":"2026-08-15T19:45:41.709487Z","submitted_at":"2025-05-31T13:26:51Z","title":"Score Matching With Missing Data","version":1},"cited_work":{"arxiv_id":"2506.00557","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.00557","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Juraj Gottweis, Wei-Hung Weng, Alexander Daryin, Tao Tu, Anil Palepu, Petar Sirkovic, Artiom Myaskovsky, Felix Weissenberger, Keran Rong, Ryutaro Tanno, et al","venue":null,"work_id":"ba776978-122d-42d7-b7c5-e18b1288beab","year":2025},"citing_paper":{"arxiv_id":"2605.16616","last_updated":"2026-05-15T20:35:32Z","snapshot_observed_at":"2026-08-16T06:46:32.334621Z","submitted_at":"2026-05-15T20:35:32Z","title":"MLReplicate: Benchmarking Autonomous Research Systems for Machine Learning Reproducibility","version":1},"reference_index":11,"source":"pdf_text","source_observed_at":"2026-05-20T19:59:40.519962Z"},"links":{"cited_paper":"/paper/2506.00557","citing_paper":"/paper/2605.16616"},"observation_digest":"sha256:7c12a43452e2891bff3f7f1f781d5a783bc181a429e6e7fe12d9756343810554","observation_id":"9925bb4d-f02a-4681-90b2-6c30b7d84333","resolution":{"observed_at":"2026-05-20T20:03:44.124752Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.18864","last_updated":"2025-02-26T06:17:13Z","snapshot_observed_at":"2026-07-06T02:11:23.670680Z","submitted_at":"2025-02-26T06:17:13Z","title":"Towards an AI co-scientist","version":1},"cited_work":{"arxiv_id":"2502.18864","doi":"10.3917/res.232.0099","metadata_source":"pith","pith_arxiv_id":"2502.18864","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Towards an AI co-scientist","venue":"cs.AI","work_id":"485486b1-a1a2-4cde-bdda-768930c403e6","year":2025},"citing_paper":{"arxiv_id":"2605.16616","last_updated":"2026-05-15T20:35:32Z","snapshot_observed_at":"2026-08-16T06:46:32.334621Z","submitted_at":"2026-05-15T20:35:32Z","title":"MLReplicate: Benchmarking Autonomous Research Systems for Machine Learning Reproducibility","version":1},"reference_index":12,"source":"pdf_text","source_observed_at":"2026-05-20T19:59:40.519962Z"},"links":{"cited_paper":"/paper/2502.18864","citing_paper":"/paper/2605.16616"},"observation_digest":"sha256:3fb37bfe11bbe91fbef7f30dcd5ee02f28443a15f5d777b5862eb21eb9664c0e","observation_id":"e027d510-67ac-46a0-b34e-d04081890451","resolution":{"observed_at":"2026-05-20T20:03:44.108786Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.02314","last_updated":"2025-06-02T23:04:12Z","snapshot_observed_at":"2026-08-18T17:40:27.870689Z","submitted_at":"2025-06-02T23:04:12Z","title":"ResearchCodeBench: Benchmarking LLMs on Implementing Novel Machine Learning Research Code","version":1},"cited_work":{"arxiv_id":"2506.02314","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.02314","snapshot_observed_at":"2026-06-29T12:23:24.671544Z","title":"Truong, Weixin Liang, Fan-Yun Sun, and Nick Haber","venue":null,"work_id":"87d25587-298e-4185-97fa-87cc76a87d92","year":2025},"citing_paper":{"arxiv_id":"2605.16616","last_updated":"2026-05-15T20:35:32Z","snapshot_observed_at":"2026-08-16T06:46:32.334621Z","submitted_at":"2026-05-15T20:35:32Z","title":"MLReplicate: Benchmarking Autonomous Research Systems for Machine Learning Reproducibility","version":1},"reference_index":13,"source":"pdf_text","source_observed_at":"2026-05-20T19:59:40.519962Z"},"links":{"cited_paper":"/paper/2506.02314","citing_paper":"/paper/2605.16616"},"observation_digest":"sha256:89508ce27b811c19e9141b2b6eb89684e8a69a87d9b652ee7e02d128c2226387","observation_id":"f78992f4-4ce3-462a-b6a9-68a132b93bb6","resolution":{"observed_at":"2026-05-20T20:03:44.142766Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2510.24891","last_updated":"2026-07-16T03:28:48Z","snapshot_observed_at":"2026-08-13T19:37:28.564290Z","submitted_at":"2025-10-28T18:54:51Z","title":"Idea2Plan: Exploring AI-Powered Research Planning","version":2},"cited_work":{"arxiv_id":"2510.24891","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2510.24891","snapshot_observed_at":"2026-07-17T02:20:35.737828Z","title":"Qian Huang, Jian V ora, Percy Liang, and Jure Leskovec","venue":null,"work_id":"be80f32d-aeca-4921-8309-857be5b75850","year":2025},"citing_paper":{"arxiv_id":"2605.16616","last_updated":"2026-05-15T20:35:32Z","snapshot_observed_at":"2026-08-16T06:46:32.334621Z","submitted_at":"2026-05-15T20:35:32Z","title":"MLReplicate: Benchmarking Autonomous Research Systems for Machine Learning Reproducibility","version":1},"reference_index":14,"source":"pdf_text","source_observed_at":"2026-05-20T19:59:40.519962Z"},"links":{"cited_paper":"/paper/2510.24891","citing_paper":"/paper/2605.16616"},"observation_digest":"sha256:ca490e8a7a615ebbaefcd6dfe55167ecaa16f4630de11c87d6c78f5488b6f29b","observation_id":"b69a8c44-cb0f-4d33-b023-26623db79095","resolution":{"observed_at":"2026-07-17T02:20:35.737828Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2310.03302","last_updated":"2024-04-14T21:02:16Z","snapshot_observed_at":"2026-08-17T14:56:37.479640Z","submitted_at":"2023-10-05T04:06:12Z","title":"MLAgentBench: Evaluating Language Agents on Machine Learning Experimentation","version":2},"cited_work":{"arxiv_id":"2310.03302","doi":null,"metadata_source":"pith","pith_arxiv_id":"2310.03302","snapshot_observed_at":"2026-07-05T17:51:14.743354Z","title":"Mlagentbench: Evaluating language agents on ma- chine learning experimentation","venue":"cs.LG","work_id":"5655b20a-fbb7-4a39-8605-d6e1d689895a","year":2023},"citing_paper":{"arxiv_id":"2605.16616","last_updated":"2026-05-15T20:35:32Z","snapshot_observed_at":"2026-08-16T06:46:32.334621Z","submitted_at":"2026-05-15T20:35:32Z","title":"MLReplicate: Benchmarking Autonomous Research Systems for Machine Learning Reproducibility","version":1},"reference_index":15,"source":"pdf_text","source_observed_at":"2026-05-20T19:59:40.519962Z"},"links":{"cited_paper":"/paper/2310.03302","citing_paper":"/paper/2605.16616"},"observation_digest":"sha256:53a724d95b74f3fa25083d6702419dbef0219c2ff9fd45e34d8cb2931759fd94","observation_id":"faa7c95a-d138-42c4-95fb-b654ee60a4e1","resolution":{"observed_at":"2026-05-20T20:03:43.962049Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"InProceedings of the 2025 Conference of the Nations of the Americas Chapter of the Association for Computational Linguistics: Human Language Technologies (System Demonstrations)","venue":null,"work_id":"941210ea-c2ce-414f-8d76-a17222ccb735","year":2025},"citing_paper":{"arxiv_id":"2605.16616","last_updated":"2026-05-15T20:35:32Z","snapshot_observed_at":"2026-08-16T06:46:32.334621Z","submitted_at":"2026-05-15T20:35:32Z","title":"MLReplicate: Benchmarking Autonomous Research Systems for Machine Learning Reproducibility","version":1},"reference_index":16,"source":"pdf_text","source_observed_at":"2026-05-20T19:59:40.519962Z"},"links":{"citing_paper":"/paper/2605.16616"},"observation_digest":"sha256:6bd41f8b5d4d3c508e9dcd8deab80ffcbeb4e0561573c25e1bee26fbf3ac1778","observation_id":"1bd1277b-9871-45f5-a3f5-18f0af05c50e","resolution":{"observed_at":"2026-05-20T20:03:59.427903Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"https://www.intology.ai/blog/ zochi-tech-report","venue":null,"work_id":"d827fa38-ba7f-4ef7-ac95-c27cbb28fba1","year":2025},"citing_paper":{"arxiv_id":"2605.16616","last_updated":"2026-05-15T20:35:32Z","snapshot_observed_at":"2026-08-16T06:46:32.334621Z","submitted_at":"2026-05-15T20:35:32Z","title":"MLReplicate: Benchmarking Autonomous Research Systems for Machine Learning Reproducibility","version":1},"reference_index":17,"source":"pdf_text","source_observed_at":"2026-05-20T19:59:40.519962Z"},"links":{"citing_paper":"/paper/2605.16616"},"observation_digest":"sha256:a9f89ab3679168cb963e4b5faa7e7865c27a5341fd624052f94adfee1f43057f","observation_id":"9551d984-19d5-4aa2-a7df-748a34ed964b","resolution":{"observed_at":"2026-05-20T20:03:59.431562Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.13138","last_updated":"2025-02-18T18:57:21Z","snapshot_observed_at":"2026-08-05T07:03:00.655237Z","submitted_at":"2025-02-18T18:57:21Z","title":"AIDE: AI-Driven Exploration in the Space of Code","version":1},"cited_work":{"arxiv_id":"2502.13138","doi":"10.48550/arxiv.2502.13138","metadata_source":"pith","pith_arxiv_id":"2502.13138","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"AIDE: AI-Driven Exploration in the Space of Code","venue":"cs.AI","work_id":"22aa3d2a-9edd-44c4-b8b8-1442ea805e01","year":2025},"citing_paper":{"arxiv_id":"2605.16616","last_updated":"2026-05-15T20:35:32Z","snapshot_observed_at":"2026-08-16T06:46:32.334621Z","submitted_at":"2026-05-15T20:35:32Z","title":"MLReplicate: Benchmarking Autonomous Research Systems for Machine Learning Reproducibility","version":1},"reference_index":18,"source":"pdf_text","source_observed_at":"2026-05-20T19:59:40.519962Z"},"links":{"cited_paper":"/paper/2502.13138","citing_paper":"/paper/2605.16616"},"observation_digest":"sha256:c4c1b62da27b81cd4ca5056118c38e9ca065cd1fd96c4e9584ed57e3a060c643","observation_id":"e9e1b2ee-6db9-40a2-8b14-06cb12a01656","resolution":{"observed_at":"2026-05-20T20:03:43.996922Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-05-25T19:23:24.934132+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-25T19:23:24.934132+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Machine Learning115, 5 (2026)","venue":null,"work_id":"6b6073af-b9f9-40cf-9238-46aea373e7b3","year":2026},"citing_paper":{"arxiv_id":"2605.16616","last_updated":"2026-05-15T20:35:32Z","snapshot_observed_at":"2026-08-16T06:46:32.334621Z","submitted_at":"2026-05-15T20:35:32Z","title":"MLReplicate: Benchmarking Autonomous Research Systems for Machine Learning Reproducibility","version":1},"reference_index":19,"source":"pdf_text","source_observed_at":"2026-05-20T19:59:40.519962Z"},"links":{"citing_paper":"/paper/2605.16616"},"observation_digest":"sha256:832e3f2401d9acde1097444a09c32c2a0e2a9935bd92b1e423b2bab1518022ce","observation_id":"8eabc3ba-713e-42fd-99c0-7591fd6dfcd7","resolution":{"observed_at":"2026-05-20T20:03:59.410566Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.04966","last_updated":"2025-05-08T05:51:48Z","snapshot_observed_at":"2026-08-19T12:35:41.643565Z","submitted_at":"2025-05-08T05:51:48Z","title":"Position: The AI Conference Peer Review Crisis Demands Author Feedback and Reviewer Rewards","version":1},"cited_work":{"arxiv_id":"2505.04966","doi":"10.48550/arxiv.2505.04966","metadata_source":"arxiv_reference","pith_arxiv_id":"2505.04966","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Position: The ai conference peer review crisis demands author feedback and reviewer rewards.arXiv preprint arXiv:2505.04966","venue":"ArXiv.org","work_id":"d194954f-8398-4908-b210-5a436774c741","year":2025},"citing_paper":{"arxiv_id":"2605.16616","last_updated":"2026-05-15T20:35:32Z","snapshot_observed_at":"2026-08-16T06:46:32.334621Z","submitted_at":"2026-05-15T20:35:32Z","title":"MLReplicate: Benchmarking Autonomous Research Systems for Machine Learning Reproducibility","version":1},"reference_index":20,"source":"pdf_text","source_observed_at":"2026-05-20T19:59:40.519962Z"},"links":{"cited_paper":"/paper/2505.04966","citing_paper":"/paper/2605.16616"},"observation_digest":"sha256:9455f5c9fba1c7b1ae822e8e211f8e57a1713f157a2f5f40e620f43a8f00899c","observation_id":"c9f67b27-7dd0-4b06-a420-6ccc2042afe6","resolution":{"observed_at":"2026-05-20T20:03:44.025076Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-08T03:38:05.942028+00:00","source":"crossref_status_cache"},{"observed_at":"2026-08-08T03:38:05.942028+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.16069","last_updated":"2025-02-26T02:33:28Z","snapshot_observed_at":"2026-08-20T18:06:48.926070Z","submitted_at":"2025-02-22T03:58:19Z","title":"Curie: Toward Rigorous and Automated Scientific Experimentation with AI Agents","version":2},"cited_work":{"arxiv_id":"2502.16069","doi":"10.48550/arxiv.2502.16069","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.16069","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Curie: Toward rigorous and automated scientific experimentation with ai agents","venue":"ArXiv.org","work_id":"ec1e1f7c-6395-4622-83ca-50df300c7a50","year":2025},"citing_paper":{"arxiv_id":"2605.16616","last_updated":"2026-05-15T20:35:32Z","snapshot_observed_at":"2026-08-16T06:46:32.334621Z","submitted_at":"2026-05-15T20:35:32Z","title":"MLReplicate: Benchmarking Autonomous Research Systems for Machine Learning Reproducibility","version":1},"reference_index":21,"source":"pdf_text","source_observed_at":"2026-05-20T19:59:40.519962Z"},"links":{"cited_paper":"/paper/2502.16069","citing_paper":"/paper/2605.16616"},"observation_digest":"sha256:1c2dd87ba2ab196aeecd0068607ba46cfda986a42344a12d6f4d449ae7e34591","observation_id":"9b55ba47-63d8-44a0-baa9-3762a2d948e9","resolution":{"observed_at":"2026-05-20T20:03:44.031083Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2408.14033","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-07-03T20:48:56.436734Z","title":"Mlr-copilot: Autonomous machine learning research based on large language models agents","venue":null,"work_id":"db834380-b5e5-4e2a-be67-c5e65541799e","year":2024},"citing_paper":{"arxiv_id":"2605.16616","last_updated":"2026-05-15T20:35:32Z","snapshot_observed_at":"2026-08-16T06:46:32.334621Z","submitted_at":"2026-05-15T20:35:32Z","title":"MLReplicate: Benchmarking Autonomous Research Systems for Machine Learning Reproducibility","version":1},"reference_index":22,"source":"pdf_text","source_observed_at":"2026-05-20T19:59:40.519962Z"},"links":{"citing_paper":"/paper/2605.16616"},"observation_digest":"sha256:04fd523b823fba286ff560b9f87447e54be2fdabec5de98b5858c4adf059cdcc","observation_id":"0b27f761-0e83-4968-b7d1-d84806b97925","resolution":{"observed_at":"2026-05-20T20:03:43.991684Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2408.06292","last_updated":"2024-09-01T00:41:18Z","snapshot_observed_at":"2026-08-20T13:44:59.356502Z","submitted_at":"2024-08-12T16:58:11Z","title":"The AI Scientist: Towards Fully Automated Open-Ended Scientific Discovery","version":3},"cited_work":{"arxiv_id":"2408.06292","doi":"10.48550/arxiv.2408.06292","metadata_source":"pith","pith_arxiv_id":"2408.06292","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"The AI Scientist: Towards Fully Automated Open-Ended Scientific Discovery","venue":"cs.AI","work_id":"56b6b58d-e73a-4317-896e-36ac5f84e957","year":2024},"citing_paper":{"arxiv_id":"2605.16616","last_updated":"2026-05-15T20:35:32Z","snapshot_observed_at":"2026-08-16T06:46:32.334621Z","submitted_at":"2026-05-15T20:35:32Z","title":"MLReplicate: Benchmarking Autonomous Research Systems for Machine Learning Reproducibility","version":1},"reference_index":23,"source":"pdf_text","source_observed_at":"2026-05-20T19:59:40.519962Z"},"links":{"cited_paper":"/paper/2408.06292","citing_paper":"/paper/2605.16616"},"observation_digest":"sha256:91931117def5ba285225a0b1de17485e3d6a652dac7341f8d87d6f758398bc21","observation_id":"8ece0086-ce39-4559-a478-f253f869ddb4","resolution":{"observed_at":"2026-05-20T20:03:44.036698Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-05-25T19:23:20.611352+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-25T19:23:20.611352+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.15266","last_updated":"2025-08-28T19:09:57Z","snapshot_observed_at":"2026-08-18T18:57:22.523654Z","submitted_at":"2025-04-21T17:47:46Z","title":"Roll the dice & look before you leap: Going beyond the creative limits of next-token prediction","version":4},"cited_work":{"arxiv_id":"2504.15266","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2504.15266","snapshot_observed_at":"2026-07-01T22:16:17.003825Z","title":"Roll the dice & look before you leap: Going beyond the creative limits of next-token prediction.arXiv preprint arXiv:2504.15266","venue":null,"work_id":"be971d12-fe47-4e91-862d-d0342692a83b","year":2025},"citing_paper":{"arxiv_id":"2605.16616","last_updated":"2026-05-15T20:35:32Z","snapshot_observed_at":"2026-08-16T06:46:32.334621Z","submitted_at":"2026-05-15T20:35:32Z","title":"MLReplicate: Benchmarking Autonomous Research Systems for Machine Learning Reproducibility","version":1},"reference_index":24,"source":"pdf_text","source_observed_at":"2026-05-20T19:59:40.519962Z"},"links":{"cited_paper":"/paper/2504.15266","citing_paper":"/paper/2605.16616"},"observation_digest":"sha256:c9ffb2bb8f7b08d4632e301a07e3e5d0c6f874abdde0b5c890a698ae573f64a7","observation_id":"300413d4-f31d-4a8e-9275-467a4e57aec8","resolution":{"observed_at":"2026-05-20T20:03:44.013482Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.14499","last_updated":"2025-02-20T12:28:23Z","snapshot_observed_at":"2026-08-17T16:58:05.353739Z","submitted_at":"2025-02-20T12:28:23Z","title":"MLGym: A New Framework and Benchmark for Advancing AI Research Agents","version":1},"cited_work":{"arxiv_id":"2502.14499","doi":"10.48550/arxiv.2502.14499","metadata_source":"arxiv_reference","pith_arxiv_id":"2502.14499","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Mlgym: A new framework and benchmark for advancing ai research agents","venue":"ArXiv.org","work_id":"0b9c0950-4be9-4a42-a77f-55c8e4a80b1e","year":2025},"citing_paper":{"arxiv_id":"2605.16616","last_updated":"2026-05-15T20:35:32Z","snapshot_observed_at":"2026-08-16T06:46:32.334621Z","submitted_at":"2026-05-15T20:35:32Z","title":"MLReplicate: Benchmarking Autonomous Research Systems for Machine Learning Reproducibility","version":1},"reference_index":25,"source":"pdf_text","source_observed_at":"2026-05-20T19:59:40.519962Z"},"links":{"cited_paper":"/paper/2502.14499","citing_paper":"/paper/2605.16616"},"observation_digest":"sha256:936a9f62ac20309b0884c89c1a32f5a72e960232195944fc77a45ee8688e2970","observation_id":"bb6bf2b1-c166-4a23-b57c-4b04936e025b","resolution":{"observed_at":"2026-05-20T20:03:44.085167Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.14191","last_updated":"2025-05-24T02:40:54Z","snapshot_observed_at":"2026-08-18T19:38:07.784057Z","submitted_at":"2025-04-19T05:35:45Z","title":"AI Idea Bench 2025: AI Research Idea Generation Benchmark","version":3},"cited_work":{"arxiv_id":"2504.14191","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2504.14191","snapshot_observed_at":"2026-06-28T18:42:28.930343Z","title":"Gollam Rabby, Diyana Muhammed, Prasenjit Mitra, and Sören Auer","venue":null,"work_id":"6fbf9394-7bfc-45d1-9d00-234e848b5ac5","year":2025},"citing_paper":{"arxiv_id":"2605.16616","last_updated":"2026-05-15T20:35:32Z","snapshot_observed_at":"2026-08-16T06:46:32.334621Z","submitted_at":"2026-05-15T20:35:32Z","title":"MLReplicate: Benchmarking Autonomous Research Systems for Machine Learning Reproducibility","version":1},"reference_index":26,"source":"pdf_text","source_observed_at":"2026-05-20T19:59:40.519962Z"},"links":{"cited_paper":"/paper/2504.14191","citing_paper":"/paper/2605.16616"},"observation_digest":"sha256:e4cf3f497cd380bcb66e155c0764c546d016dac1ea4b06d061be5b42f965271c","observation_id":"71594fa6-0b39-467d-a124-ec2601e58981","resolution":{"observed_at":"2026-05-20T20:03:43.974081Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2503.19309","last_updated":"2025-03-25T03:14:53Z","snapshot_observed_at":"2026-08-19T03:29:24.092990Z","submitted_at":"2025-03-25T03:14:53Z","title":"Iterative Hypothesis Generation for Scientific Discovery with Monte Carlo Nash Equilibrium Self-Refining Trees","version":1},"cited_work":{"arxiv_id":"2503.19309","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2503.19309","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Samuel Schmidgall, Yusheng Su, Ze Wang, Ximeng Sun, Jialian Wu, Xiaodong Yu, Jiang Liu, Michael Moor, Zicheng Liu, and Emad Barsoum","venue":null,"work_id":"73980325-c625-4305-89fa-428c8bebec32","year":2025},"citing_paper":{"arxiv_id":"2605.16616","last_updated":"2026-05-15T20:35:32Z","snapshot_observed_at":"2026-08-16T06:46:32.334621Z","submitted_at":"2026-05-15T20:35:32Z","title":"MLReplicate: Benchmarking Autonomous Research Systems for Machine Learning Reproducibility","version":1},"reference_index":27,"source":"pdf_text","source_observed_at":"2026-05-20T19:59:40.519962Z"},"links":{"cited_paper":"/paper/2503.19309","citing_paper":"/paper/2605.16616"},"observation_digest":"sha256:28893f5fd144ad3bacb5d3695efb81c508c2e66ab44d7b7401479c8b1d41051a","observation_id":"0756e5a5-7897-4a1e-b28d-37b3d15f0a15","resolution":{"observed_at":"2026-05-20T20:03:43.967487Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Minju Seo, Jinheon Baek, Seongyun Lee, and Sung Ju Hwang","venue":null,"work_id":"fdcb985f-71c1-41ef-97f5-03336fd77fe4","year":2025},"citing_paper":{"arxiv_id":"2605.16616","last_updated":"2026-05-15T20:35:32Z","snapshot_observed_at":"2026-08-16T06:46:32.334621Z","submitted_at":"2026-05-15T20:35:32Z","title":"MLReplicate: Benchmarking Autonomous Research Systems for Machine Learning Reproducibility","version":1},"reference_index":28,"source":"pdf_text","source_observed_at":"2026-05-20T19:59:40.519962Z"},"links":{"citing_paper":"/paper/2605.16616"},"observation_digest":"sha256:aeb0fe97f8a65a8cba8b201f7084219614970626cc179ba79bc3d97cdd20b1ce","observation_id":"b3a48133-c903-433c-a2f9-319a0e231ab7","resolution":{"observed_at":"2026-05-20T20:03:59.423970Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":"2504.17192","doi":"10.48550/arxiv.2504.17192","metadata_source":"arxiv_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Zachary S Siegel, Sayash Kapoor, Nitya Nagdir, Benedikt Stroebl, and Arvind Narayanan","venue":"ArXiv.org","work_id":"90d70472-2cc8-421e-85af-7f4f74d1bc7e","year":2025},"citing_paper":{"arxiv_id":"2605.16616","last_updated":"2026-05-15T20:35:32Z","snapshot_observed_at":"2026-08-16T06:46:32.334621Z","submitted_at":"2026-05-15T20:35:32Z","title":"MLReplicate: Benchmarking Autonomous Research Systems for Machine Learning Reproducibility","version":1},"reference_index":29,"source":"pdf_text","source_observed_at":"2026-05-20T19:59:40.519962Z"},"links":{"citing_paper":"/paper/2605.16616"},"observation_digest":"sha256:114f7f98035b6c7c5072f5033d87cbf4a0884287dd626f11888454daf25e51ad","observation_id":"5a6e87bc-99c4-4d28-829b-d371a70de1b4","resolution":{"observed_at":"2026-05-20T20:03:44.074843Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2409.11363","last_updated":"2026-06-22T22:36:12Z","snapshot_observed_at":"2026-08-16T13:17:45.152269Z","submitted_at":"2024-09-17T17:13:19Z","title":"CORE-Bench: Fostering the Credibility of Published Research Through a Computational Reproducibility Agent Benchmark","version":2},"cited_work":{"arxiv_id":"2409.11363","doi":null,"metadata_source":"pith","pith_arxiv_id":"2409.11363","snapshot_observed_at":"2026-07-04T17:20:00.275604Z","title":"Core-bench: Fos- tering the credibility of published research through a computational reproducibility agent benchmark","venue":"cs.CL","work_id":"cdbaede6-c7c7-40e2-8758-2781354ec391","year":2024},"citing_paper":{"arxiv_id":"2605.16616","last_updated":"2026-05-15T20:35:32Z","snapshot_observed_at":"2026-08-16T06:46:32.334621Z","submitted_at":"2026-05-15T20:35:32Z","title":"MLReplicate: Benchmarking Autonomous Research Systems for Machine Learning Reproducibility","version":1},"reference_index":30,"source":"pdf_text","source_observed_at":"2026-05-20T19:59:40.519962Z"},"links":{"cited_paper":"/paper/2409.11363","citing_paper":"/paper/2605.16616"},"observation_digest":"sha256:a0b8ce8207c0275ce0a2ae40e9c9db1f2720dcc89d78a2c08b57812a93adafcf","observation_id":"0e380be1-f4b2-4628-a75c-6e3b81f58de2","resolution":{"observed_at":"2026-06-24T01:14:19.028324Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.13228","last_updated":"2025-06-11T15:39:13Z","snapshot_observed_at":"2026-08-18T22:23:52.701008Z","submitted_at":"2025-02-18T19:06:21Z","title":"Conformal Prediction as Bayesian Quadrature","version":2},"cited_work":{"arxiv_id":"2502.13228","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2502.13228","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Giulio Starace, Oliver Jaffe, Dane Sherburn, James Aung, Jun Shern Chan, Leon Maksin, Rachel Dias, Evan Mays, Benjamin Kinsella, Wyatt Thompson, et al","venue":null,"work_id":"fb721c46-1fb4-465b-b763-622a72e4a825","year":2025},"citing_paper":{"arxiv_id":"2605.16616","last_updated":"2026-05-15T20:35:32Z","snapshot_observed_at":"2026-08-16T06:46:32.334621Z","submitted_at":"2026-05-15T20:35:32Z","title":"MLReplicate: Benchmarking Autonomous Research Systems for Machine Learning Reproducibility","version":1},"reference_index":31,"source":"pdf_text","source_observed_at":"2026-05-20T19:59:40.519962Z"},"links":{"cited_paper":"/paper/2502.13228","citing_paper":"/paper/2605.16616"},"observation_digest":"sha256:61c049ddb7bf578bbb1314cf19976f57bfcb9522259ee3d462ed14be9839c2d3","observation_id":"deb8ea53-0f27-4d6b-8000-42b9fc854511","resolution":{"observed_at":"2026-05-20T20:03:44.007931Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.01848","last_updated":"2025-04-07T12:15:49Z","snapshot_observed_at":"2026-07-06T21:03:06.857885Z","submitted_at":"2025-04-02T15:55:24Z","title":"PaperBench: Evaluating AI's Ability to Replicate AI Research","version":3},"cited_work":{"arxiv_id":"2504.01848","doi":"10.1126/science.adp7429","metadata_source":"pith","pith_arxiv_id":"2504.01848","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"PaperBench: Evaluating AI's Ability to Replicate AI Research","venue":"cs.AI","work_id":"7db23490-3dd0-4cfa-a75c-5537ee36dcc2","year":2025},"citing_paper":{"arxiv_id":"2605.16616","last_updated":"2026-05-15T20:35:32Z","snapshot_observed_at":"2026-08-16T06:46:32.334621Z","submitted_at":"2026-05-15T20:35:32Z","title":"MLReplicate: Benchmarking Autonomous Research Systems for Machine Learning Reproducibility","version":1},"reference_index":32,"source":"pdf_text","source_observed_at":"2026-05-20T19:59:40.519962Z"},"links":{"cited_paper":"/paper/2504.01848","citing_paper":"/paper/2605.16616"},"observation_digest":"sha256:55ae51dcadb749d5c179492154d918e86d6b06cd57df5e01f7abfaa5bc234874","observation_id":"f52f1028-c800-40ec-a0c4-f19b8a7aee55","resolution":{"observed_at":"2026-05-20T20:03:44.002213Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2505.18705","last_updated":"2025-05-24T13:54:38Z","snapshot_observed_at":"2026-08-20T00:11:39.023419Z","submitted_at":"2025-05-24T13:54:38Z","title":"AI-Researcher: Autonomous Scientific Innovation","version":1},"cited_work":{"arxiv_id":"2505.18705","doi":"10.48550/arxiv.2505.18705","metadata_source":"pith","pith_arxiv_id":"2505.18705","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"CoRR , volume =","venue":"cs.AI","work_id":"3845f0f0-08d4-4650-b390-6bfdd269f79a","year":2025},"citing_paper":{"arxiv_id":"2605.16616","last_updated":"2026-05-15T20:35:32Z","snapshot_observed_at":"2026-08-16T06:46:32.334621Z","submitted_at":"2026-05-15T20:35:32Z","title":"MLReplicate: Benchmarking Autonomous Research Systems for Machine Learning Reproducibility","version":1},"reference_index":33,"source":"pdf_text","source_observed_at":"2026-05-20T19:59:40.519962Z"},"links":{"cited_paper":"/paper/2505.18705","citing_paper":"/paper/2605.16616"},"observation_digest":"sha256:1e872ad2b6286cf651c2735a536cd29881d0667237afda961c0a36685c2b2f71","observation_id":"3e720c4e-45a8-4000-ac09-a942e9132d31","resolution":{"observed_at":"2026-05-20T20:03:44.047519Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-05-25T19:23:26.377836+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-25T19:23:26.377836+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2312.11805","last_updated":"2025-05-09T21:04:06Z","snapshot_observed_at":"2026-08-20T18:27:04.837880Z","submitted_at":"2023-12-19T02:39:27Z","title":"Gemini: A Family of Highly Capable Multimodal Models","version":5},"cited_work":{"arxiv_id":"2312.11805","doi":"10.1038/nrn2888","metadata_source":"pith","pith_arxiv_id":"2312.11805","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Gemini: A Family of Highly Capable Multimodal Models","venue":"cs.CL","work_id":"83f7c85b-3f11-450f-ac0c-64d9745220b2","year":2023},"citing_paper":{"arxiv_id":"2605.16616","last_updated":"2026-05-15T20:35:32Z","snapshot_observed_at":"2026-08-16T06:46:32.334621Z","submitted_at":"2026-05-15T20:35:32Z","title":"MLReplicate: Benchmarking Autonomous Research Systems for Machine Learning Reproducibility","version":1},"reference_index":34,"source":"pdf_text","source_observed_at":"2026-05-20T19:59:40.519962Z"},"links":{"cited_paper":"/paper/2312.11805","citing_paper":"/paper/2605.16616"},"observation_digest":"sha256:e8926c71321ea6c523af3dd5899774af0ba070da5230ff9119ef08c2a502674f","observation_id":"d3fa7a11-82fd-420e-a2f6-a7b46370ccf0","resolution":{"observed_at":"2026-05-20T20:03:44.041705Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Yixuan Weng, Minjun Zhu, Guangsheng Bao, Hongbo Zhang, Jindong Wang, Yue Zhang, and Linyi Yang","venue":null,"work_id":"7a8880af-0bab-48ee-b948-c2f39ed230a4","year":2023},"citing_paper":{"arxiv_id":"2605.16616","last_updated":"2026-05-15T20:35:32Z","snapshot_observed_at":"2026-08-16T06:46:32.334621Z","submitted_at":"2026-05-15T20:35:32Z","title":"MLReplicate: Benchmarking Autonomous Research Systems for Machine Learning Reproducibility","version":1},"reference_index":35,"source":"pdf_text","source_observed_at":"2026-05-20T19:59:40.519962Z"},"links":{"citing_paper":"/paper/2605.16616"},"observation_digest":"sha256:a81ce3659feb240baad61bbf19558042286b379e5718db77170d70d5169c81c5","observation_id":"2cc68a4f-c023-4249-8f2b-22f56d670f2d","resolution":{"observed_at":"2026-05-20T20:03:59.413939Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2411.00816","last_updated":"2025-03-08T14:01:34Z","snapshot_observed_at":"2026-08-17T08:54:33.605275Z","submitted_at":"2024-10-28T08:10:21Z","title":"CycleResearcher: Improving Automated Research via Automated Review","version":3},"cited_work":{"arxiv_id":"2411.00816","doi":"10.48550/arxiv.2411.00816","metadata_source":"arxiv_reference","pith_arxiv_id":"2411.00816","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"Shirley Wu, Michel Galley, Baolin Peng, Hao Cheng, Gavin Li, Yao Dou, Weixin Cai, James Zou, Jure Leskovec, and Jianfeng Gao","venue":"arXiv (Cornell University)","work_id":"529c57f4-8402-4221-93f2-032f08d64085","year":2025},"citing_paper":{"arxiv_id":"2605.16616","last_updated":"2026-05-15T20:35:32Z","snapshot_observed_at":"2026-08-16T06:46:32.334621Z","submitted_at":"2026-05-15T20:35:32Z","title":"MLReplicate: Benchmarking Autonomous Research Systems for Machine Learning Reproducibility","version":1},"reference_index":36,"source":"pdf_text","source_observed_at":"2026-05-20T19:59:40.519962Z"},"links":{"cited_paper":"/paper/2411.00816","citing_paper":"/paper/2605.16616"},"observation_digest":"sha256:7f8fd8e5b283fe6909ee9bbc8410a95dd257bb07247809548ba0a9766d7dcd3e","observation_id":"cedccc7d-ee4a-4bdb-97c2-a53372c98bab","resolution":{"observed_at":"2026-05-20T20:03:44.069785Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2502.00640","last_updated":"2025-07-29T22:56:43Z","snapshot_observed_at":"2026-08-17T17:04:08.332548Z","submitted_at":"2025-02-02T03:05:52Z","title":"CollabLLM: From Passive Responders to Active Collaborators","version":3},"cited_work":{"arxiv_id":"2502.00640","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2502.00640","snapshot_observed_at":"2026-07-02T13:16:58.901104Z","title":"arXiv preprint arXiv:2502.00640(2025)","venue":null,"work_id":"a58f7a6a-6769-464d-bb3f-66c4db72dbec","year":2025},"citing_paper":{"arxiv_id":"2605.16616","last_updated":"2026-05-15T20:35:32Z","snapshot_observed_at":"2026-08-16T06:46:32.334621Z","submitted_at":"2026-05-15T20:35:32Z","title":"MLReplicate: Benchmarking Autonomous Research Systems for Machine Learning Reproducibility","version":1},"reference_index":37,"source":"pdf_text","source_observed_at":"2026-05-20T19:59:40.519962Z"},"links":{"cited_paper":"/paper/2502.00640","citing_paper":"/paper/2605.16616"},"observation_digest":"sha256:ce63ae909ffd8398ba582fec42eabc678e2e5a2a105d0e7a821da7bc3c0b8d16","observation_id":"a28acc7c-07de-4b87-8203-a5a354d290f8","resolution":{"observed_at":"2026-05-20T20:03:44.080101Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.00255","last_updated":"2025-08-07T16:31:54Z","snapshot_observed_at":"2026-08-16T12:45:09.699136Z","submitted_at":"2025-03-31T22:02:24Z","title":"SciReplicate-Bench: Benchmarking LLMs in Agent-driven Algorithmic Reproduction from Research Papers","version":2},"cited_work":{"arxiv_id":"2504.00255","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2504.00255","snapshot_observed_at":"2026-07-03T13:08:07.735939Z","title":"Scireplicate-bench: Benchmarking llms in agent-driven algorithmic reproduction from research papers","venue":null,"work_id":"7dded0d9-b6a4-4d10-baaf-2ea3d9551127","year":2025},"citing_paper":{"arxiv_id":"2605.16616","last_updated":"2026-05-15T20:35:32Z","snapshot_observed_at":"2026-08-16T06:46:32.334621Z","submitted_at":"2026-05-15T20:35:32Z","title":"MLReplicate: Benchmarking Autonomous Research Systems for Machine Learning Reproducibility","version":1},"reference_index":38,"source":"pdf_text","source_observed_at":"2026-05-20T19:59:40.519962Z"},"links":{"cited_paper":"/paper/2504.00255","citing_paper":"/paper/2605.16616"},"observation_digest":"sha256:780ae811f32682ea531eeb571bbec561ad1f61b9b694b82f6ebc41d3449e03c1","observation_id":"f49d13c1-7280-445b-b733-8e2cccab9811","resolution":{"observed_at":"2026-05-20T20:03:44.090179Z","resolver_source":"arxiv_id","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2604.26506","last_updated":"2026-04-29T10:11:12Z","snapshot_observed_at":"2026-08-15T15:41:15.828801Z","submitted_at":"2026-04-29T10:11:12Z","title":"SafeReview: Defending LLM-based Review Systems Against Adversarial Hidden Prompts","version":1},"cited_work":{"arxiv_id":"2604.26506","doi":null,"metadata_source":"pith","pith_arxiv_id":"2604.26506","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"SafeReview: Defending LLM-based Review Systems Against Adversarial Hidden Prompts","venue":"cs.CL","work_id":"1251910b-d335-47c4-8782-f8fd0c97a139","year":2026},"citing_paper":{"arxiv_id":"2605.16616","last_updated":"2026-05-15T20:35:32Z","snapshot_observed_at":"2026-08-16T06:46:32.334621Z","submitted_at":"2026-05-15T20:35:32Z","title":"MLReplicate: Benchmarking Autonomous Research Systems for Machine Learning Reproducibility","version":1},"reference_index":39,"source":"pdf_text","source_observed_at":"2026-05-20T19:59:40.519962Z"},"links":{"cited_paper":"/paper/2604.26506","citing_paper":"/paper/2605.16616"},"observation_digest":"sha256:cdf3bec387469fa80eb23ae9a2f7f9f8a79c39129ad81a5d90b92ec4e8a8ed6e","observation_id":"7ec5ac8a-d6b0-46f5-b3da-3512722da8d8","resolution":{"observed_at":"2026-05-20T20:03:44.052307Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2504.08066","last_updated":"2025-04-10T18:44:41Z","snapshot_observed_at":"2026-08-10T03:18:36.275195Z","submitted_at":"2025-04-10T18:44:41Z","title":"The AI Scientist-v2: Workshop-Level Automated Scientific Discovery via Agentic Tree Search","version":1},"cited_work":{"arxiv_id":"2504.08066","doi":"10.48550/arxiv.2504.08066","metadata_source":"pith","pith_arxiv_id":"2504.08066","snapshot_observed_at":"2026-08-05T02:28:24.338817Z","title":"The AI Scientist-v2: Workshop-Level Automated Scientific Discovery via Agentic Tree Search","venue":"cs.AI","work_id":"fa04f346-ee20-4e9d-bf04-3ad3569a8ed1","year":2025},"citing_paper":{"arxiv_id":"2605.16616","last_updated":"2026-05-15T20:35:32Z","snapshot_observed_at":"2026-08-16T06:46:32.334621Z","submitted_at":"2026-05-15T20:35:32Z","title":"MLReplicate: Benchmarking Autonomous Research Systems for Machine Learning Reproducibility","version":1},"reference_index":40,"source":"pdf_text","source_observed_at":"2026-05-20T19:59:40.519962Z"},"links":{"cited_paper":"/paper/2504.08066","citing_paper":"/paper/2605.16616"},"observation_digest":"sha256:271df38ca15feb8f51fb9e8b8b8d70c28b270628f447a0ad34bebcc6d4b670e6","observation_id":"9aff99f0-4850-4eb6-b1f0-dde0fc3ad75f","resolution":{"observed_at":"2026-05-20T20:03:44.057404Z","resolver_source":"local_arxiv","status":"metadata_mismatch"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-05-25T19:23:27.206897+00:00","source":"crossref_status_cache"},{"observed_at":"2026-05-25T19:23:27.206897+00:00","source":"openalex_status_cache"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":{"arxiv_id":"2506.01372","last_updated":"2025-06-09T09:01:24Z","snapshot_observed_at":"2026-08-15T19:53:32.389209Z","submitted_at":"2025-06-02T06:59:10Z","title":"AI Scientists Fail Without Strong Implementation Capability","version":2},"cited_work":{"arxiv_id":"2506.01372","doi":null,"metadata_source":"arxiv_reference","pith_arxiv_id":"2506.01372","snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"Evaluation conducted exclusively on synthetic datasets","venue":null,"work_id":"0b2c1af8-3ba1-48eb-b5ba-301f7b61d917","year":2025},"citing_paper":{"arxiv_id":"2605.16616","last_updated":"2026-05-15T20:35:32Z","snapshot_observed_at":"2026-08-16T06:46:32.334621Z","submitted_at":"2026-05-15T20:35:32Z","title":"MLReplicate: Benchmarking Autonomous Research Systems for Machine Learning Reproducibility","version":1},"reference_index":41,"source":"pdf_text","source_observed_at":"2026-05-20T19:59:40.519962Z"},"links":{"cited_paper":"/paper/2506.01372","citing_paper":"/paper/2605.16616"},"observation_digest":"sha256:9e0994824788b2cb3caafdddcc7d9ae314814e2877215665dcbc03369607887d","observation_id":"f88f855f-8dd1-40e0-8d56-2bd040ecf91c","resolution":{"observed_at":"2026-05-20T20:03:44.063892Z","resolver_source":"arxiv_id","status":"verified_exact"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"experiment_name","venue":null,"work_id":"796d8358-2d41-40d0-90cd-33c6a94ff07e","year":2025},"citing_paper":{"arxiv_id":"2605.16616","last_updated":"2026-05-15T20:35:32Z","snapshot_observed_at":"2026-08-16T06:46:32.334621Z","submitted_at":"2026-05-15T20:35:32Z","title":"MLReplicate: Benchmarking Autonomous Research Systems for Machine Learning Reproducibility","version":1},"reference_index":42,"source":"pdf_text","source_observed_at":"2026-05-20T19:59:40.519962Z"},"links":{"citing_paper":"/paper/2605.16616"},"observation_digest":"sha256:ea6d164610b71ea53231a50c68eb4e3dd0cc212fae68a28b1535bce06205b435","observation_id":"5f89d451-e4d1-40b6-bac6-7eecae9d90ae","resolution":{"observed_at":"2026-05-20T20:03:59.417577Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}},{"citation":{"cited_paper":null,"cited_work":{"arxiv_id":null,"doi":null,"metadata_source":"raw_reference","pith_arxiv_id":null,"snapshot_observed_at":"2026-06-05T21:23:00.469572Z","title":"target\"","venue":null,"work_id":"25d74a19-9dcb-4a7e-b62c-8102495a465a","year":2025},"citing_paper":{"arxiv_id":"2605.16616","last_updated":"2026-05-15T20:35:32Z","snapshot_observed_at":"2026-08-16T06:46:32.334621Z","submitted_at":"2026-05-15T20:35:32Z","title":"MLReplicate: Benchmarking Autonomous Research Systems for Machine Learning Reproducibility","version":1},"reference_index":43,"source":"pdf_text","source_observed_at":"2026-05-20T19:59:40.519962Z"},"links":{"citing_paper":"/paper/2605.16616"},"observation_digest":"sha256:9916a80fc5c06ccf342599a0c066ca2878c9562bd2bc4e8e5f62b18355138d33","observation_id":"41d5decd-2981-446f-815a-19628d7fe986","resolution":{"observed_at":"2026-05-20T20:03:59.420844Z","resolver_source":"raw_fallback","status":"verified_fuzzy"},"standing_notice":{"events":[],"observation":"No event found in the named queried sources as of 2026-08-22T06:32:14.747728+00:00.","reason":null,"source_receipts":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"state":"measured"}}],"paper":{"arxiv_id":"2605.16616","last_updated":"2026-05-15T20:35:32Z","latest_version":1,"primary_category":"cs.LG","snapshot_observed_at":"2026-08-16T06:46:32.334621Z","submitted_at":"2026-05-15T20:35:32Z","title":"MLReplicate: Benchmarking Autonomous Research Systems for Machine Learning Reproducibility"},"reference_resolution":{"displayed":43,"state_counts":{"malformed_identifier":0,"metadata_mismatch":19,"parse_uncertain":0,"unresolved":0,"verified_exact":13,"verified_fuzzy":11},"total_outbound_references":43},"refusal":"A citation records a reference. It does not transfer a finding from one paper to another.","schema":"pith.paper-citation-record.v1","standing_sources":[{"observed_at":"2026-08-22T06:32:14.747728+00:00","source":"crossref"},{"observed_at":"2026-08-22T06:32:06.552537+00:00","source":"retraction_watch"}],"thesis":"As of 22 August 2026, this Paper Citation Record lists 43 of 43 outbound references and 1 inbound Pith citation observation for arXiv:2605.16616."}