{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2026:PMIYGK5JXCWBO2MRMMLA2NUH3T","short_pith_number":"pith:PMIYGK5J","schema_version":"1.0","canonical_sha256":"7b11832ba9b8ac17699163160d3687dcde1cca8de72b687b4cd1f0971a9f7ec9","source":{"kind":"arxiv","id":"2607.27191","version":1},"attestation_state":"computed","paper":{"title":"Can AI agents conduct open-ended AI research? Early evidence from two case studies","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CY","cs.LG"],"primary_cat":"cs.AI","authors_text":"Abhishek Shetty, Andrew Schwartz, Arvind Narayanan, Cozmin Ududec, David Africa, Derrick Chan-Sew, Gillian Hadfield, Harry Coppock, Helen Toner, Konstantinos Voudouris, Magda Dubois, Matilda Orona, Nitya Nadgir, Peter Kirgis, Rishi Bommasani, Sayash Kapoor, Seth Lazar, Shoshannah Tekofsky, Stephan Rabanser, Steve Newman, Tilman Bayer, Toby Pilditch, Viet Nguyen, Yue Ling","submitted_at":"2026-07-29T17:57:19Z","abstract_excerpt":"Forecasts of explosive AI progress hinge on AI agents automating AI research. But evidence on whether agents can carry out open-ended AI research is thin. Current evaluations either test agents on narrow, verifiable tasks, which excludes open-ended research, or submit AI-generated papers to blind peer review, which is overstretched, stochastic, and suffers from poor review quality. We introduce a third way to measure progress towards AI R\\&D automation. An agent takes on the central, open-ended research question of a high-quality unpublished paper, and the paper's original authors grade its ou"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2607.27191","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2026-07-29T17:57:19Z","cross_cats_sorted":["cs.CY","cs.LG"],"title_canon_sha256":"6b2cfca672465ed880e97e877b8b65086e85616e8504ae41e63cbe5c2bc645dc","abstract_canon_sha256":"cb87e3edcf833944a519e0047b707e26148cf8bc6137dadbec5daf489e981138"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"7b11832ba9b8ac17699163160d3687dcde1cca8de72b687b4cd1f0971a9f7ec9","last_reissued_at":"2026-07-30T01:24:04.221560Z","signature_status":"unsigned_v0","first_computed_at":"2026-07-30T01:24:04.221560Z"},"graph_snapshot":{"paper":{"title":"Can AI agents conduct open-ended AI research? Early evidence from two case studies","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CY","cs.LG"],"primary_cat":"cs.AI","authors_text":"Abhishek Shetty, Andrew Schwartz, Arvind Narayanan, Cozmin Ududec, David Africa, Derrick Chan-Sew, Gillian Hadfield, Harry Coppock, Helen Toner, Konstantinos Voudouris, Magda Dubois, Matilda Orona, Nitya Nadgir, Peter Kirgis, Rishi Bommasani, Sayash Kapoor, Seth Lazar, Shoshannah Tekofsky, Stephan Rabanser, Steve Newman, Tilman Bayer, Toby Pilditch, Viet Nguyen, Yue Ling","submitted_at":"2026-07-29T17:57:19Z","abstract_excerpt":"Forecasts of explosive AI progress hinge on AI agents automating AI research. But evidence on whether agents can carry out open-ended AI research is thin. Current evaluations either test agents on narrow, verifiable tasks, which excludes open-ended research, or submit AI-generated papers to blind peer review, which is overstretched, stochastic, and suffers from poor review quality. We introduce a third way to measure progress towards AI R\\&D automation. An agent takes on the central, open-ended research question of a high-quality unpublished paper, and the paper's original authors grade its ou"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2607.27191","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2607.27191/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2607.27191","created_at":"2026-07-30T01:24:04.227056+00:00"},{"alias_kind":"arxiv_version","alias_value":"2607.27191v1","created_at":"2026-07-30T01:24:04.227056+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2607.27191","created_at":"2026-07-30T01:24:04.227056+00:00"},{"alias_kind":"pith_short_12","alias_value":"PMIYGK5JXCWB","created_at":"2026-07-30T01:24:04.227056+00:00"},{"alias_kind":"pith_short_16","alias_value":"PMIYGK5JXCWBO2MR","created_at":"2026-07-30T01:24:04.227056+00:00"},{"alias_kind":"pith_short_8","alias_value":"PMIYGK5J","created_at":"2026-07-30T01:24:04.227056+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/PMIYGK5JXCWBO2MRMMLA2NUH3T","json":"https://pith.science/pith/PMIYGK5JXCWBO2MRMMLA2NUH3T.json","graph_json":"https://pith.science/api/pith-number/PMIYGK5JXCWBO2MRMMLA2NUH3T/graph.json","events_json":"https://pith.science/api/pith-number/PMIYGK5JXCWBO2MRMMLA2NUH3T/events.json","paper":"https://pith.science/paper/PMIYGK5J"},"agent_actions":{"view_html":"https://pith.science/pith/PMIYGK5JXCWBO2MRMMLA2NUH3T","download_json":"https://pith.science/pith/PMIYGK5JXCWBO2MRMMLA2NUH3T.json","view_paper":"https://pith.science/paper/PMIYGK5J","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2607.27191&json=true","fetch_graph":"https://pith.science/api/pith-number/PMIYGK5JXCWBO2MRMMLA2NUH3T/graph.json","fetch_events":"https://pith.science/api/pith-number/PMIYGK5JXCWBO2MRMMLA2NUH3T/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/PMIYGK5JXCWBO2MRMMLA2NUH3T/action/timestamp_anchor","attest_storage":"https://pith.science/pith/PMIYGK5JXCWBO2MRMMLA2NUH3T/action/storage_attestation","attest_author":"https://pith.science/pith/PMIYGK5JXCWBO2MRMMLA2NUH3T/action/author_attestation","sign_citation":"https://pith.science/pith/PMIYGK5JXCWBO2MRMMLA2NUH3T/action/citation_signature","submit_replication":"https://pith.science/pith/PMIYGK5JXCWBO2MRMMLA2NUH3T/action/replication_record"}},"created_at":"2026-07-30T01:24:04.227056+00:00","updated_at":"2026-07-30T01:24:04.227056+00:00"}