{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:5JVY6J7V72XT665XRADOVIJUBZ","short_pith_number":"pith:5JVY6J7V","schema_version":"1.0","canonical_sha256":"ea6b8f27f5feaf3f7bb78806eaa1340e41c39e671395c3d5ea30348083958e57","source":{"kind":"arxiv","id":"2409.04109","version":1},"attestation_state":"computed","paper":{"title":"Can LLMs Generate Novel Research Ideas? A Large-Scale Human Study with 100+ NLP Researchers","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CY","cs.HC","cs.LG"],"primary_cat":"cs.CL","authors_text":"Chenglei Si, Diyi Yang, Tatsunori Hashimoto","submitted_at":"2024-09-06T08:25:03Z","abstract_excerpt":"Recent advancements in large language models (LLMs) have sparked optimism about their potential to accelerate scientific discovery, with a growing number of works proposing research agents that autonomously generate and validate new ideas. Despite this, no evaluations have shown that LLM systems can take the very first step of producing novel, expert-level ideas, let alone perform the entire research process. We address this by establishing an experimental design that evaluates research idea generation while controlling for confounders and performs the first head-to-head comparison between exp"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2409.04109","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-09-06T08:25:03Z","cross_cats_sorted":["cs.AI","cs.CY","cs.HC","cs.LG"],"title_canon_sha256":"bc6f2331ba6b41ceab1c596dd0708f196b50b499e21b9afdefaae21da5b4c9c4","abstract_canon_sha256":"36aa9e5ba58fb14d3c26b6f5d75182f02c8c5bf75a9fe09b900c6774e76eeaad"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:04:01.634715Z","signature_b64":"a1gFopP5tXDPANIN572qSd8wPHJN5+P6HBzO1VvjIv1WVVXuHAtPXt/MN8wKinhilzo60RNb4rghDbyNIMeLDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ea6b8f27f5feaf3f7bb78806eaa1340e41c39e671395c3d5ea30348083958e57","last_reissued_at":"2026-07-05T09:04:01.634303Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:04:01.634303Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Can LLMs Generate Novel Research Ideas? A Large-Scale Human Study with 100+ NLP Researchers","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CY","cs.HC","cs.LG"],"primary_cat":"cs.CL","authors_text":"Chenglei Si, Diyi Yang, Tatsunori Hashimoto","submitted_at":"2024-09-06T08:25:03Z","abstract_excerpt":"Recent advancements in large language models (LLMs) have sparked optimism about their potential to accelerate scientific discovery, with a growing number of works proposing research agents that autonomously generate and validate new ideas. Despite this, no evaluations have shown that LLM systems can take the very first step of producing novel, expert-level ideas, let alone perform the entire research process. We address this by establishing an experimental design that evaluates research idea generation while controlling for confounders and performs the first head-to-head comparison between exp"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2409.04109","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2409.04109/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2409.04109","created_at":"2026-07-05T09:04:01.634378+00:00"},{"alias_kind":"arxiv_version","alias_value":"2409.04109v1","created_at":"2026-07-05T09:04:01.634378+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2409.04109","created_at":"2026-07-05T09:04:01.634378+00:00"},{"alias_kind":"pith_short_12","alias_value":"5JVY6J7V72XT","created_at":"2026-07-05T09:04:01.634378+00:00"},{"alias_kind":"pith_short_16","alias_value":"5JVY6J7V72XT665X","created_at":"2026-07-05T09:04:01.634378+00:00"},{"alias_kind":"pith_short_8","alias_value":"5JVY6J7V","created_at":"2026-07-05T09:04:01.634378+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":33,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.08758","citing_title":"Ideas Have Genomes: Benchmarking Scientific Lineage Reasoning and Lineage-Grounded Idea Generation","ref_index":37,"is_internal_anchor":true},{"citing_arxiv_id":"2606.22610","citing_title":"PaperClaw: Harnessing Agents for Autonomous Research and Human-in-the-Loop Refinement","ref_index":60,"is_internal_anchor":false},{"citing_arxiv_id":"2606.21228","citing_title":"Sakana Fugu Technical Report","ref_index":152,"is_internal_anchor":false},{"citing_arxiv_id":"2606.11337","citing_title":"Can AI Agents Synthesize Scientific Conclusions?","ref_index":107,"is_internal_anchor":false},{"citing_arxiv_id":"2606.08234","citing_title":"SciTrace: Trajectory-Aware Safety Reasoning for Scientific Discovery Agents","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2606.31651","citing_title":"FARS: A Fully Automated Research System Deployed at Scale","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2605.14790","citing_title":"Graphs of Research: Citation Evolution Graphs as Supervision for Research Idea Generation","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2606.29630","citing_title":"SFBench: The SciFy Scientific Feasibility Benchmark","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2605.26396","citing_title":"Advancing Creative Physical Intelligence in Large Multimodal Models","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2606.08723","citing_title":"From Text to Discovery: How Large Language Models Are Reshaping Research Across Scientific and Humanistic Disciplines","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2409.14634","citing_title":"Human-LLM Compound System for Scientific Ideation through Facet Recombination and Novelty Evaluation","ref_index":66,"is_internal_anchor":false},{"citing_arxiv_id":"2605.21491","citing_title":"Teaching Language Models to Forecast Research Success Through Comparative Idea Evaluation","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2507.21035","citing_title":"GenoMAS: A Multi-Agent Framework for Scientific Discovery via Code-Driven Gene Expression Analysis","ref_index":101,"is_internal_anchor":false},{"citing_arxiv_id":"2603.01092","citing_title":"The Alien Space of Science: Sampling Coherent but Cognitively Unavailable Research Directions","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18661","citing_title":"AI for Auto-Research: Roadmap & User Guide","ref_index":184,"is_internal_anchor":false},{"citing_arxiv_id":"2605.19156","citing_title":"How Far Are We From True Auto-Research?","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2506.22598","citing_title":"RExBench: Can coding agents autonomously implement AI research extensions?","ref_index":38,"is_internal_anchor":false},{"citing_arxiv_id":"2510.01553","citing_title":"IoDResearch: Deep Research on Private Heterogeneous Data via the Internet of Data","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2601.10038","citing_title":"What Understanding Means in AI-Laden Astronomy","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2601.10038","citing_title":"What Understanding Means in AI-Laden Astronomy","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2509.19349","citing_title":"ShinkaEvolve: Towards Open-Ended And Sample-Efficient Program Evolution","ref_index":116,"is_internal_anchor":false},{"citing_arxiv_id":"2601.15170","citing_title":"Multi-Dimensional Knowledge Profiling with Large-Scale Literature Database and Hierarchical Retrieval","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2604.09597","citing_title":"From Theory to Protocol: Executable Frameworks for Creative Emergence and Strategic Foresight","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2605.13450","citing_title":"Assessing the Creativity of Large Language Models: Testing, Limits, and New Frontiers","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2604.03338","citing_title":"The Ideation Bottleneck: Decomposing the Quality Gap Between AI-Generated and Human Economics Research","ref_index":2,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/5JVY6J7V72XT665XRADOVIJUBZ","json":"https://pith.science/pith/5JVY6J7V72XT665XRADOVIJUBZ.json","graph_json":"https://pith.science/api/pith-number/5JVY6J7V72XT665XRADOVIJUBZ/graph.json","events_json":"https://pith.science/api/pith-number/5JVY6J7V72XT665XRADOVIJUBZ/events.json","paper":"https://pith.science/paper/5JVY6J7V"},"agent_actions":{"view_html":"https://pith.science/pith/5JVY6J7V72XT665XRADOVIJUBZ","download_json":"https://pith.science/pith/5JVY6J7V72XT665XRADOVIJUBZ.json","view_paper":"https://pith.science/paper/5JVY6J7V","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2409.04109&json=true","fetch_graph":"https://pith.science/api/pith-number/5JVY6J7V72XT665XRADOVIJUBZ/graph.json","fetch_events":"https://pith.science/api/pith-number/5JVY6J7V72XT665XRADOVIJUBZ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/5JVY6J7V72XT665XRADOVIJUBZ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/5JVY6J7V72XT665XRADOVIJUBZ/action/storage_attestation","attest_author":"https://pith.science/pith/5JVY6J7V72XT665XRADOVIJUBZ/action/author_attestation","sign_citation":"https://pith.science/pith/5JVY6J7V72XT665XRADOVIJUBZ/action/citation_signature","submit_replication":"https://pith.science/pith/5JVY6J7V72XT665XRADOVIJUBZ/action/replication_record"}},"created_at":"2026-07-05T09:04:01.634378+00:00","updated_at":"2026-07-05T09:04:01.634378+00:00"}