{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:CMFLGHS7SBDTHUIGHAYFXRVNTG","short_pith_number":"pith:CMFLGHS7","schema_version":"1.0","canonical_sha256":"130ab31e5f904733d10638305bc6ad99b570c23a6da0baec61ac14d630d5acbb","source":{"kind":"arxiv","id":"2506.00794","version":1},"attestation_state":"computed","paper":{"title":"Predicting Empirical AI Research Outcomes with Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Chenglei Si, He He, Jiaxin Wen, Shi Feng, Yueh-han Chen","submitted_at":"2025-06-01T02:46:31Z","abstract_excerpt":"Many promising-looking ideas in AI research fail to deliver, but their validation takes substantial human labor and compute. Predicting an idea's chance of success is thus crucial for accelerating empirical AI research, a skill that even expert researchers can only acquire through substantial experience. We build the first benchmark for this task and compare LMs with human experts. Concretely, given two research ideas (e.g., two jailbreaking methods), we aim to predict which will perform better on a set of benchmarks. We scrape ideas and experimental results from conference papers, yielding 1,"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2506.00794","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2025-06-01T02:46:31Z","cross_cats_sorted":[],"title_canon_sha256":"87a6e33d9d7e0596221d618569aba8c86c430c8b6bd8c04f20688c76b054b32b","abstract_canon_sha256":"1969f4c5346c9d9eba1d71ab09591e60fb0e42a5bb46bc59e63585a91a07d0b7"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:13:40.560569Z","signature_b64":"V2kepj8NvvQ4POIY1hI/fBFDsfQl/QugWeo9G8NZtuD6KN+DP0ylhD9o7VeuTccQPZ9owL0yQ3Kfh3LvNyZOBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"130ab31e5f904733d10638305bc6ad99b570c23a6da0baec61ac14d630d5acbb","last_reissued_at":"2026-07-05T11:13:40.560104Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:13:40.560104Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Predicting Empirical AI Research Outcomes with Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Chenglei Si, He He, Jiaxin Wen, Shi Feng, Yueh-han Chen","submitted_at":"2025-06-01T02:46:31Z","abstract_excerpt":"Many promising-looking ideas in AI research fail to deliver, but their validation takes substantial human labor and compute. Predicting an idea's chance of success is thus crucial for accelerating empirical AI research, a skill that even expert researchers can only acquire through substantial experience. We build the first benchmark for this task and compare LMs with human experts. Concretely, given two research ideas (e.g., two jailbreaking methods), we aim to predict which will perform better on a set of benchmarks. We scrape ideas and experimental results from conference papers, yielding 1,"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.00794","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2506.00794/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2506.00794","created_at":"2026-07-05T11:13:40.560159+00:00"},{"alias_kind":"arxiv_version","alias_value":"2506.00794v1","created_at":"2026-07-05T11:13:40.560159+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.00794","created_at":"2026-07-05T11:13:40.560159+00:00"},{"alias_kind":"pith_short_12","alias_value":"CMFLGHS7SBDT","created_at":"2026-07-05T11:13:40.560159+00:00"},{"alias_kind":"pith_short_16","alias_value":"CMFLGHS7SBDTHUIG","created_at":"2026-07-05T11:13:40.560159+00:00"},{"alias_kind":"pith_short_8","alias_value":"CMFLGHS7","created_at":"2026-07-05T11:13:40.560159+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2604.23593","citing_title":"When AI reviews science: Can we trust the referee?","ref_index":102,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/CMFLGHS7SBDTHUIGHAYFXRVNTG","json":"https://pith.science/pith/CMFLGHS7SBDTHUIGHAYFXRVNTG.json","graph_json":"https://pith.science/api/pith-number/CMFLGHS7SBDTHUIGHAYFXRVNTG/graph.json","events_json":"https://pith.science/api/pith-number/CMFLGHS7SBDTHUIGHAYFXRVNTG/events.json","paper":"https://pith.science/paper/CMFLGHS7"},"agent_actions":{"view_html":"https://pith.science/pith/CMFLGHS7SBDTHUIGHAYFXRVNTG","download_json":"https://pith.science/pith/CMFLGHS7SBDTHUIGHAYFXRVNTG.json","view_paper":"https://pith.science/paper/CMFLGHS7","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2506.00794&json=true","fetch_graph":"https://pith.science/api/pith-number/CMFLGHS7SBDTHUIGHAYFXRVNTG/graph.json","fetch_events":"https://pith.science/api/pith-number/CMFLGHS7SBDTHUIGHAYFXRVNTG/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/CMFLGHS7SBDTHUIGHAYFXRVNTG/action/timestamp_anchor","attest_storage":"https://pith.science/pith/CMFLGHS7SBDTHUIGHAYFXRVNTG/action/storage_attestation","attest_author":"https://pith.science/pith/CMFLGHS7SBDTHUIGHAYFXRVNTG/action/author_attestation","sign_citation":"https://pith.science/pith/CMFLGHS7SBDTHUIGHAYFXRVNTG/action/citation_signature","submit_replication":"https://pith.science/pith/CMFLGHS7SBDTHUIGHAYFXRVNTG/action/replication_record"}},"created_at":"2026-07-05T11:13:40.560159+00:00","updated_at":"2026-07-05T11:13:40.560159+00:00"}