{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:MJGHZXSXS3IPWFH5FSEPU7QTX6","short_pith_number":"pith:MJGHZXSX","schema_version":"1.0","canonical_sha256":"624c7cde5796d0fb14fd2c88fa7e13bf9c6d5338e8fa263f9ddf2a13d5eb4167","source":{"kind":"arxiv","id":"2502.14499","version":1},"attestation_state":"computed","paper":{"title":"MLGym: A New Framework and Benchmark for Advancing AI Research Agents","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Ajay Menon, Amar Budhiraja, Deepak Nathani, Despoina Magka, Dieuwke Hupkes, Gaurav Chaurasia, Jakob Foerster, Lovish Madaan, Nicholas Roberts, Nikolay Bashlykov, Ricardo Silveira Cabral, Roberta Raileanu, Tatiana Shavrina, Vincent Moens, Vladislav Vorotilov, William Yang Wang, Yoram Bachrach","submitted_at":"2025-02-20T12:28:23Z","abstract_excerpt":"We introduce Meta MLGym and MLGym-Bench, a new framework and benchmark for evaluating and developing LLM agents on AI research tasks. This is the first Gym environment for machine learning (ML) tasks, enabling research on reinforcement learning (RL) algorithms for training such agents. MLGym-bench consists of 13 diverse and open-ended AI research tasks from diverse domains such as computer vision, natural language processing, reinforcement learning, and game theory. Solving these tasks requires real-world AI research skills such as generating new ideas and hypotheses, creating and processing d"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.14499","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.CL","submitted_at":"2025-02-20T12:28:23Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"0f395d8dd4ae9816faaac03a57ca605626afc8e4d08981125a3cc59a372e6087","abstract_canon_sha256":"d9bae0fe9a624ee177e886fb45722c25d04fa44a0740feb0ea125bef04e79e44"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:17:27.601764Z","signature_b64":"vMXWRSdmOn4dTnYalieyL8J8nPoxLNHiJ22/ouS6xoRcnc3RJPTxbVPXdHDxH/I4scKTUaFLQXobLgKUeFoMAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"624c7cde5796d0fb14fd2c88fa7e13bf9c6d5338e8fa263f9ddf2a13d5eb4167","last_reissued_at":"2026-07-05T10:17:27.601283Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:17:27.601283Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"MLGym: A New Framework and Benchmark for Advancing AI Research Agents","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Ajay Menon, Amar Budhiraja, Deepak Nathani, Despoina Magka, Dieuwke Hupkes, Gaurav Chaurasia, Jakob Foerster, Lovish Madaan, Nicholas Roberts, Nikolay Bashlykov, Ricardo Silveira Cabral, Roberta Raileanu, Tatiana Shavrina, Vincent Moens, Vladislav Vorotilov, William Yang Wang, Yoram Bachrach","submitted_at":"2025-02-20T12:28:23Z","abstract_excerpt":"We introduce Meta MLGym and MLGym-Bench, a new framework and benchmark for evaluating and developing LLM agents on AI research tasks. This is the first Gym environment for machine learning (ML) tasks, enabling research on reinforcement learning (RL) algorithms for training such agents. MLGym-bench consists of 13 diverse and open-ended AI research tasks from diverse domains such as computer vision, natural language processing, reinforcement learning, and game theory. Solving these tasks requires real-world AI research skills such as generating new ideas and hypotheses, creating and processing d"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.14499","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.14499/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.14499","created_at":"2026-07-05T10:17:27.601361+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.14499v1","created_at":"2026-07-05T10:17:27.601361+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.14499","created_at":"2026-07-05T10:17:27.601361+00:00"},{"alias_kind":"pith_short_12","alias_value":"MJGHZXSXS3IP","created_at":"2026-07-05T10:17:27.601361+00:00"},{"alias_kind":"pith_short_16","alias_value":"MJGHZXSXS3IPWFH5","created_at":"2026-07-05T10:17:27.601361+00:00"},{"alias_kind":"pith_short_8","alias_value":"MJGHZXSX","created_at":"2026-07-05T10:17:27.601361+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":22,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.24530","citing_title":"NatureBench: Can Coding Agents Match the Published SOTA of Nature-Family Papers?","ref_index":64,"is_internal_anchor":false},{"citing_arxiv_id":"2606.22866","citing_title":"Discovering Crystal Structure Prediction Algorithms with an AI Co-Scientist","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2606.07591","citing_title":"ResearchClawBench: A Benchmark for End-to-End Autonomous Scientific Research","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2606.13148","citing_title":"TerraBench: Can Agents Reason Over Heterogeneous Earth-System Data?","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2606.12736","citing_title":"Benchmarking AI Agents for Addressing Scientific Challenges Across Scales","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2606.09550","citing_title":"InquiTree: Evaluating AI Agents in the Scientific Inquiry Loop with Paper-Derived Research Trees","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2606.13148","citing_title":"TerraBench: Can Agents Reason Over Heterogeneous Earth-System Data?","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2606.03544","citing_title":"SAGE: A Quantitative Evaluation of Socialized Evolution in Agent Ecosystems","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08678","citing_title":"MLS-Bench: A Holistic and Rigorous Assessment of AI Systems on Building Better AI","ref_index":69,"is_internal_anchor":false},{"citing_arxiv_id":"2606.07591","citing_title":"ResearchClawBench: A Benchmark for End-to-End Autonomous Scientific Research","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2606.17838","citing_title":"Environment-Grounded Automated Prompt Optimization for LLM Game Agents","ref_index":36,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07021","citing_title":"Behavior Cue Reasoning: Monitorable Reasoning Improves Efficiency and Safety through Oversight","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2605.16616","citing_title":"MLReplicate: Benchmarking Autonomous Research Systems for Machine Learning Reproducibility","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18661","citing_title":"AI for Auto-Research: Roadmap & User Guide","ref_index":135,"is_internal_anchor":false},{"citing_arxiv_id":"2508.07407","citing_title":"A Comprehensive Survey of Self-Evolving AI Agents: A New Paradigm Bridging Foundation Models and Lifelong Agentic Systems","ref_index":62,"is_internal_anchor":false},{"citing_arxiv_id":"2504.19678","citing_title":"From LLM Reasoning to Autonomous AI Agents: A Comprehensive Review","ref_index":122,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08678","citing_title":"MLS-Bench: A Holistic and Rigorous Assessment of AI Systems on Building Better AI","ref_index":67,"is_internal_anchor":false},{"citing_arxiv_id":"2604.14116","citing_title":"TREX: Automating LLM Fine-tuning via Agent-Driven Tree-based Exploration","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2604.06566","citing_title":"AI-Driven Research for Databases","ref_index":58,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07021","citing_title":"Behavior Cue Reasoning: Monitorable Reasoning Improves Efficiency and Safety through Oversight","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2604.06111","citing_title":"AgentCE-Bench: Agent Configurable Evaluation with Scalable Horizons and Controllable Difficulty under Lightweight Environments","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01250","citing_title":"EO-Gym: A Multimodal, Interactive Environment for Earth Observation Agents","ref_index":31,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/MJGHZXSXS3IPWFH5FSEPU7QTX6","json":"https://pith.science/pith/MJGHZXSXS3IPWFH5FSEPU7QTX6.json","graph_json":"https://pith.science/api/pith-number/MJGHZXSXS3IPWFH5FSEPU7QTX6/graph.json","events_json":"https://pith.science/api/pith-number/MJGHZXSXS3IPWFH5FSEPU7QTX6/events.json","paper":"https://pith.science/paper/MJGHZXSX"},"agent_actions":{"view_html":"https://pith.science/pith/MJGHZXSXS3IPWFH5FSEPU7QTX6","download_json":"https://pith.science/pith/MJGHZXSXS3IPWFH5FSEPU7QTX6.json","view_paper":"https://pith.science/paper/MJGHZXSX","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.14499&json=true","fetch_graph":"https://pith.science/api/pith-number/MJGHZXSXS3IPWFH5FSEPU7QTX6/graph.json","fetch_events":"https://pith.science/api/pith-number/MJGHZXSXS3IPWFH5FSEPU7QTX6/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/MJGHZXSXS3IPWFH5FSEPU7QTX6/action/timestamp_anchor","attest_storage":"https://pith.science/pith/MJGHZXSXS3IPWFH5FSEPU7QTX6/action/storage_attestation","attest_author":"https://pith.science/pith/MJGHZXSXS3IPWFH5FSEPU7QTX6/action/author_attestation","sign_citation":"https://pith.science/pith/MJGHZXSXS3IPWFH5FSEPU7QTX6/action/citation_signature","submit_replication":"https://pith.science/pith/MJGHZXSXS3IPWFH5FSEPU7QTX6/action/replication_record"}},"created_at":"2026-07-05T10:17:27.601361+00:00","updated_at":"2026-07-05T10:17:27.601361+00:00"}