{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:2QME45N5QSLR7SBJGENJK3HZNF","short_pith_number":"pith:2QME45N5","schema_version":"1.0","canonical_sha256":"d4184e75bd84971fc829311a956cf969638c774c14715747bf4e4edac586563f","source":{"kind":"arxiv","id":"2505.07782","version":1},"attestation_state":"computed","paper":{"title":"MLE-Dojo: Interactive Environments for Empowering LLM Agents in Machine Learning Engineering","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Bo Dai, Changhao Li, Chao Zhang, Dingu Sagar V K, Ian Shu-Hei Wong, Percy Liang, Rongzhi Zhang, Rushi Qiang, Sherry Yang, Yinghao Li, Yuchen Zhuang","submitted_at":"2025-05-12T17:35:43Z","abstract_excerpt":"We introduce MLE-Dojo, a Gym-style framework for systematically reinforcement learning, evaluating, and improving autonomous large language model (LLM) agents in iterative machine learning engineering (MLE) workflows. Unlike existing benchmarks that primarily rely on static datasets or single-attempt evaluations, MLE-Dojo provides an interactive environment enabling agents to iteratively experiment, debug, and refine solutions through structured feedback loops. Built upon 200+ real-world Kaggle challenges, MLE-Dojo covers diverse, open-ended MLE tasks carefully curated to reflect realistic eng"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.07782","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-05-12T17:35:43Z","cross_cats_sorted":[],"title_canon_sha256":"1d00ccd4fd90079450d20a639e74d453bf5bafe08ac194f17729302ac37428a6","abstract_canon_sha256":"35f8e8ac332368974f5ba30b24253f757db31ed300a34169b971f343d60086e9"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:01:53.272330Z","signature_b64":"A818C4v62VZj0BwGe4kzMP0WFeUu28y0JvI8COeXpJTPCwxPs9dHdq2CWpSUXxDp9wCeRtKJcnOGwo9UDnu0Cg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"d4184e75bd84971fc829311a956cf969638c774c14715747bf4e4edac586563f","last_reissued_at":"2026-07-05T11:01:53.271846Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:01:53.271846Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"MLE-Dojo: Interactive Environments for Empowering LLM Agents in Machine Learning Engineering","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Bo Dai, Changhao Li, Chao Zhang, Dingu Sagar V K, Ian Shu-Hei Wong, Percy Liang, Rongzhi Zhang, Rushi Qiang, Sherry Yang, Yinghao Li, Yuchen Zhuang","submitted_at":"2025-05-12T17:35:43Z","abstract_excerpt":"We introduce MLE-Dojo, a Gym-style framework for systematically reinforcement learning, evaluating, and improving autonomous large language model (LLM) agents in iterative machine learning engineering (MLE) workflows. Unlike existing benchmarks that primarily rely on static datasets or single-attempt evaluations, MLE-Dojo provides an interactive environment enabling agents to iteratively experiment, debug, and refine solutions through structured feedback loops. Built upon 200+ real-world Kaggle challenges, MLE-Dojo covers diverse, open-ended MLE tasks carefully curated to reflect realistic eng"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.07782","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.07782/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.07782","created_at":"2026-07-05T11:01:53.271904+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.07782v1","created_at":"2026-07-05T11:01:53.271904+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.07782","created_at":"2026-07-05T11:01:53.271904+00:00"},{"alias_kind":"pith_short_12","alias_value":"2QME45N5QSLR","created_at":"2026-07-05T11:01:53.271904+00:00"},{"alias_kind":"pith_short_16","alias_value":"2QME45N5QSLR7SBJ","created_at":"2026-07-05T11:01:53.271904+00:00"},{"alias_kind":"pith_short_8","alias_value":"2QME45N5","created_at":"2026-07-05T11:01:53.271904+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":9,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.20728","citing_title":"VTOS: Learning to Orchestrate Vision Tools by Co-Searching Solutions and Observers","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2606.12191","citing_title":"Agentic Environment Engineering for Large Language Models: A Survey of Environment Modeling, Synthesis, Evaluation, and Application","ref_index":158,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08678","citing_title":"MLS-Bench: A Holistic and Rigorous Assessment of AI Systems on Building Better AI","ref_index":79,"is_internal_anchor":false},{"citing_arxiv_id":"2606.30616","citing_title":"Scaling the Horizon, Not the Parameters: Reaching Trillion-Parameter Performance with a 35B Agent","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2606.11926","citing_title":"Toward Generalist Autonomous Research via Hypothesis-Tree Refinement","ref_index":143,"is_internal_anchor":false},{"citing_arxiv_id":"2506.05425","citing_title":"SIV-Bench: A Video Benchmark for Social Interaction Understanding and Reasoning","ref_index":37,"is_internal_anchor":false},{"citing_arxiv_id":"2602.07906","citing_title":"AceGRPO: Adaptive Curriculum Enhanced Group Relative Policy Optimization for Autonomous Machine Learning Engineering","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12913","citing_title":"Revisiting DAgger in the Era of LLM-Agents","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08678","citing_title":"MLS-Bench: A Holistic and Rigorous Assessment of AI Systems on Building Better AI","ref_index":77,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/2QME45N5QSLR7SBJGENJK3HZNF","json":"https://pith.science/pith/2QME45N5QSLR7SBJGENJK3HZNF.json","graph_json":"https://pith.science/api/pith-number/2QME45N5QSLR7SBJGENJK3HZNF/graph.json","events_json":"https://pith.science/api/pith-number/2QME45N5QSLR7SBJGENJK3HZNF/events.json","paper":"https://pith.science/paper/2QME45N5"},"agent_actions":{"view_html":"https://pith.science/pith/2QME45N5QSLR7SBJGENJK3HZNF","download_json":"https://pith.science/pith/2QME45N5QSLR7SBJGENJK3HZNF.json","view_paper":"https://pith.science/paper/2QME45N5","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.07782&json=true","fetch_graph":"https://pith.science/api/pith-number/2QME45N5QSLR7SBJGENJK3HZNF/graph.json","fetch_events":"https://pith.science/api/pith-number/2QME45N5QSLR7SBJGENJK3HZNF/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/2QME45N5QSLR7SBJGENJK3HZNF/action/timestamp_anchor","attest_storage":"https://pith.science/pith/2QME45N5QSLR7SBJGENJK3HZNF/action/storage_attestation","attest_author":"https://pith.science/pith/2QME45N5QSLR7SBJGENJK3HZNF/action/author_attestation","sign_citation":"https://pith.science/pith/2QME45N5QSLR7SBJGENJK3HZNF/action/citation_signature","submit_replication":"https://pith.science/pith/2QME45N5QSLR7SBJGENJK3HZNF/action/replication_record"}},"created_at":"2026-07-05T11:01:53.271904+00:00","updated_at":"2026-07-05T11:01:53.271904+00:00"}