{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:KFEYUAX4YJCNY4PZO256EX6E3Z","short_pith_number":"pith:KFEYUAX4","schema_version":"1.0","canonical_sha256":"51498a02fcc244dc71f976bbe25fc4de63ea59d8c44df3ab4f4085fba5294f8a","source":{"kind":"arxiv","id":"2411.15114","version":2},"attestation_state":"computed","paper":{"title":"RE-Bench: Evaluating frontier AI R&D capabilities of language model agents against human experts","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Aron Lajko, Ben West, Brian Goodrich, Elena Ericheva, Elizabeth Barnes, Hjalmar Wijk, Holden Karnofsky, Jai Dhyani, Joel Becker, Josh Clymer, Katharyn Garcia, Lawrence Chan, Lucas Sato, Maksym Taran, Megan Kinniment, Michael Chen, Neev Parikh, Nikola Jurkovic, Sami Jawhar, Seraphina Nix, Tao Lin, Thomas Broadley, William Saunders","submitted_at":"2024-11-22T18:30:46Z","abstract_excerpt":"Frontier AI safety policies highlight automation of AI research and development (R&D) by AI agents as an important capability to anticipate. However, there exist few evaluations for AI R&D capabilities, and none that are highly realistic and have a direct comparison to human performance. We introduce RE-Bench (Research Engineering Benchmark, v1), which consists of 7 challenging, open-ended ML research engineering environments and data from 71 8-hour attempts by 61 distinct human experts. We confirm that our experts make progress in the environments given 8 hours, with 82% of expert attempts ac"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2411.15114","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-11-22T18:30:46Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"9ba1249590ebf1b0f21cec01756bae0115bd19dbb4c34db30c04131bf866b035","abstract_canon_sha256":"e415f1a30f91f588556fcf190cc61137bc95c139b5df3209131ac18c3896bb97"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:10:02.659942Z","signature_b64":"mHcL+fnGk7Pir+yrLOWNDnds+UNRrYy3hV+kkLyJdr1DC/+YrFJv7GV23s518AGMTNrDErfqhr/gCKbsQ4BZBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"51498a02fcc244dc71f976bbe25fc4de63ea59d8c44df3ab4f4085fba5294f8a","last_reissued_at":"2026-07-05T11:10:02.659439Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:10:02.659439Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"RE-Bench: Evaluating frontier AI R&D capabilities of language model agents against human experts","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Aron Lajko, Ben West, Brian Goodrich, Elena Ericheva, Elizabeth Barnes, Hjalmar Wijk, Holden Karnofsky, Jai Dhyani, Joel Becker, Josh Clymer, Katharyn Garcia, Lawrence Chan, Lucas Sato, Maksym Taran, Megan Kinniment, Michael Chen, Neev Parikh, Nikola Jurkovic, Sami Jawhar, Seraphina Nix, Tao Lin, Thomas Broadley, William Saunders","submitted_at":"2024-11-22T18:30:46Z","abstract_excerpt":"Frontier AI safety policies highlight automation of AI research and development (R&D) by AI agents as an important capability to anticipate. However, there exist few evaluations for AI R&D capabilities, and none that are highly realistic and have a direct comparison to human performance. We introduce RE-Bench (Research Engineering Benchmark, v1), which consists of 7 challenging, open-ended ML research engineering environments and data from 71 8-hour attempts by 61 distinct human experts. We confirm that our experts make progress in the environments given 8 hours, with 82% of expert attempts ac"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2411.15114","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2411.15114/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2411.15114","created_at":"2026-07-05T11:10:02.659500+00:00"},{"alias_kind":"arxiv_version","alias_value":"2411.15114v2","created_at":"2026-07-05T11:10:02.659500+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2411.15114","created_at":"2026-07-05T11:10:02.659500+00:00"},{"alias_kind":"pith_short_12","alias_value":"KFEYUAX4YJCN","created_at":"2026-07-05T11:10:02.659500+00:00"},{"alias_kind":"pith_short_16","alias_value":"KFEYUAX4YJCNY4PZ","created_at":"2026-07-05T11:10:02.659500+00:00"},{"alias_kind":"pith_short_8","alias_value":"KFEYUAX4","created_at":"2026-07-05T11:10:02.659500+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":29,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.06411","citing_title":"RuBench: A Repository-Level Agentic Coding Benchmark with Natively Authored Russian Task Specifications","ref_index":20,"is_internal_anchor":true},{"citing_arxiv_id":"2606.22866","citing_title":"Discovering Crystal Structure Prediction Algorithms with an AI Co-Scientist","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2606.21891","citing_title":"Learning the ARTS of Search for Automated Discovery","ref_index":54,"is_internal_anchor":false},{"citing_arxiv_id":"2606.17799","citing_title":"Position: Coding Benchmarks Are Misaligned with Agentic Software Engineering","ref_index":48,"is_internal_anchor":false},{"citing_arxiv_id":"2606.09550","citing_title":"InquiTree: Evaluating AI Agents in the Scientific Inquiry Loop with Paper-Derived Research Trees","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2607.00913","citing_title":"Two AI Metrics Diverged: Will it Make All the Difference?","ref_index":68,"is_internal_anchor":false},{"citing_arxiv_id":"2606.04455","citing_title":"The Meta-Agent Challenge: Are Current Agents Capable of Autonomous Agent Development?","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2606.05080","citing_title":"AutoLab: Can Frontier Models Solve Long-Horizon Auto Research and Engineering Tasks?","ref_index":42,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17373","citing_title":"FML-bench: A Controlled Study of AI Research Agent Strategies from the Perspective of Search Dynamics","ref_index":57,"is_internal_anchor":false},{"citing_arxiv_id":"2606.30182","citing_title":"MirrorCode: AI can rebuild entire programs from behavior alone","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2605.27492","citing_title":"Benchmarks are Not Enough: RAMP for Runtime Assessing of Agentic Models in Production Systems","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2606.11926","citing_title":"Toward Generalist Autonomous Research via Hypothesis-Tree Refinement","ref_index":160,"is_internal_anchor":false},{"citing_arxiv_id":"2605.20744","citing_title":"Hack-Verifiable Environments: Towards Evaluating Reward Hacking at Scale","ref_index":37,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17373","citing_title":"FML-bench: A Controlled Study of AI Research Agent Strategies from the Perspective of Search Dynamics","ref_index":57,"is_internal_anchor":false},{"citing_arxiv_id":"2507.11473","citing_title":"Chain of Thought Monitorability: A New and Fragile Opportunity for AI Safety","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18661","citing_title":"AI for Auto-Research: Roadmap & User Guide","ref_index":222,"is_internal_anchor":false},{"citing_arxiv_id":"2605.19156","citing_title":"How Far Are We From True Auto-Research?","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2507.06261","citing_title":"Gemini 2.5: Pushing the Frontier with Advanced Reasoning, Multimodality, Long Context, and Next Generation Agentic Capabilities","ref_index":89,"is_internal_anchor":false},{"citing_arxiv_id":"2510.12826","citing_title":"Scheming Ability in LLM-to-LLM Strategic Interactions","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2412.04984","citing_title":"Frontier Models are Capable of In-context Scheming","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2602.09514","citing_title":"EcoGym: Evaluating LLMs for Long-Horizon Plan-and-Execute in Interactive Economies","ref_index":38,"is_internal_anchor":false},{"citing_arxiv_id":"2502.10517","citing_title":"KernelBench: Can LLMs Write Efficient GPU Kernels?","ref_index":42,"is_internal_anchor":false},{"citing_arxiv_id":"2605.13950","citing_title":"Collider-Bench: Benchmarking AI Agents with Particle Physics Analysis Reproduction","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2605.14445","citing_title":"FrontierSmith: Synthesizing Open-Ended Coding Problems at Scale","ref_index":36,"is_internal_anchor":false},{"citing_arxiv_id":"2604.24966","citing_title":"Risk Reporting for Developers' Internal AI Model Use","ref_index":48,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/KFEYUAX4YJCNY4PZO256EX6E3Z","json":"https://pith.science/pith/KFEYUAX4YJCNY4PZO256EX6E3Z.json","graph_json":"https://pith.science/api/pith-number/KFEYUAX4YJCNY4PZO256EX6E3Z/graph.json","events_json":"https://pith.science/api/pith-number/KFEYUAX4YJCNY4PZO256EX6E3Z/events.json","paper":"https://pith.science/paper/KFEYUAX4"},"agent_actions":{"view_html":"https://pith.science/pith/KFEYUAX4YJCNY4PZO256EX6E3Z","download_json":"https://pith.science/pith/KFEYUAX4YJCNY4PZO256EX6E3Z.json","view_paper":"https://pith.science/paper/KFEYUAX4","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2411.15114&json=true","fetch_graph":"https://pith.science/api/pith-number/KFEYUAX4YJCNY4PZO256EX6E3Z/graph.json","fetch_events":"https://pith.science/api/pith-number/KFEYUAX4YJCNY4PZO256EX6E3Z/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/KFEYUAX4YJCNY4PZO256EX6E3Z/action/timestamp_anchor","attest_storage":"https://pith.science/pith/KFEYUAX4YJCNY4PZO256EX6E3Z/action/storage_attestation","attest_author":"https://pith.science/pith/KFEYUAX4YJCNY4PZO256EX6E3Z/action/author_attestation","sign_citation":"https://pith.science/pith/KFEYUAX4YJCNY4PZO256EX6E3Z/action/citation_signature","submit_replication":"https://pith.science/pith/KFEYUAX4YJCNY4PZO256EX6E3Z/action/replication_record"}},"created_at":"2026-07-05T11:10:02.659500+00:00","updated_at":"2026-07-05T11:10:02.659500+00:00"}