{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:YZV72VZW4RF3CYA32WT76S4HT7","short_pith_number":"pith:YZV72VZW","schema_version":"1.0","canonical_sha256":"c66bfd5736e44bb1601bd5a7ff4b879ff39fd9be8faec7bad631924589dfecf0","source":{"kind":"arxiv","id":"2410.05080","version":3},"attestation_state":"computed","paper":{"title":"ScienceAgentBench: Toward Rigorous Assessment of Language Agents for Data-Driven Scientific Discovery","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Benjamin Burns, Boshi Wang, Botao Yu, Chen Wei, Daniel Adu-Ampratwum, Frazier N. Baker, Huan Sun, Mingyi Xue, Qianheng Zhang, Shijie Chen, Song Gao, Vishal Dey, Xia Ning, Xuhui Huang, Yifei Li, Yu Su, Yuting Ning, Zeyi Liao, Ziru Chen, Zitong Lu","submitted_at":"2024-10-07T14:33:50Z","abstract_excerpt":"The advancements of large language models (LLMs) have piqued growing interest in developing LLM-based language agents to automate scientific discovery end-to-end, which has sparked both excitement and skepticism about their true capabilities. In this work, we call for rigorous assessment of agents on individual tasks in a scientific workflow before making bold claims on end-to-end automation. To this end, we present ScienceAgentBench, a new benchmark for evaluating language agents for data-driven scientific discovery. To ensure the scientific authenticity and real-world relevance of our benchm"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.05080","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-10-07T14:33:50Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"b1a53ca8c518fb7a6f242f17bf78f3d1a8b2ca172779a99d7a95bf6eff49d05c","abstract_canon_sha256":"1db686719223d2d476d019c136f90496173651dbd87de83cd9eca894666cc2d1"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:41:49.046399Z","signature_b64":"8B75A7SB+YM63pukt2leG3OQwfQ9sKk5rrqQHXx9TPvGRlIL6D0eVNi8gnQdVw/m4O/68yvmbAJdU3okdDaBAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"c66bfd5736e44bb1601bd5a7ff4b879ff39fd9be8faec7bad631924589dfecf0","last_reissued_at":"2026-07-05T10:41:49.045900Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:41:49.045900Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"ScienceAgentBench: Toward Rigorous Assessment of Language Agents for Data-Driven Scientific Discovery","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Benjamin Burns, Boshi Wang, Botao Yu, Chen Wei, Daniel Adu-Ampratwum, Frazier N. Baker, Huan Sun, Mingyi Xue, Qianheng Zhang, Shijie Chen, Song Gao, Vishal Dey, Xia Ning, Xuhui Huang, Yifei Li, Yu Su, Yuting Ning, Zeyi Liao, Ziru Chen, Zitong Lu","submitted_at":"2024-10-07T14:33:50Z","abstract_excerpt":"The advancements of large language models (LLMs) have piqued growing interest in developing LLM-based language agents to automate scientific discovery end-to-end, which has sparked both excitement and skepticism about their true capabilities. In this work, we call for rigorous assessment of agents on individual tasks in a scientific workflow before making bold claims on end-to-end automation. To this end, we present ScienceAgentBench, a new benchmark for evaluating language agents for data-driven scientific discovery. To ensure the scientific authenticity and real-world relevance of our benchm"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.05080","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.05080/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.05080","created_at":"2026-07-05T10:41:49.045954+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.05080v3","created_at":"2026-07-05T10:41:49.045954+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.05080","created_at":"2026-07-05T10:41:49.045954+00:00"},{"alias_kind":"pith_short_12","alias_value":"YZV72VZW4RF3","created_at":"2026-07-05T10:41:49.045954+00:00"},{"alias_kind":"pith_short_16","alias_value":"YZV72VZW4RF3CYA3","created_at":"2026-07-05T10:41:49.045954+00:00"},{"alias_kind":"pith_short_8","alias_value":"YZV72VZW","created_at":"2026-07-05T10:41:49.045954+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":39,"internal_anchor_count":2,"sample":[{"citing_arxiv_id":"2607.07663","citing_title":"Recursive Self-Improvement in AI: From Bounded Self-Refinement to Autonomous Research Loops","ref_index":188,"is_internal_anchor":true},{"citing_arxiv_id":"2604.17406","citing_title":"EvoMaster: A Foundational Evolving Agent Framework for Agentic Science at Scale","ref_index":10,"is_internal_anchor":true},{"citing_arxiv_id":"2606.26346","citing_title":"How Do Tool-Augmented LLM Agents Perform on Real-World Energy Analytics Tasks?","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2606.20475","citing_title":"Marginal Advantage Accumulation for Memory-Driven Agent Self-Evolution","ref_index":39,"is_internal_anchor":false},{"citing_arxiv_id":"2606.18648","citing_title":"Deep Research in Physical Sciences: A Multi-Agent Framework and Comprehensive Benchmark","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2606.18356","citing_title":"SafeClawBench: Separating Semantic, Audit-Evidence, and Sandbox Harm in Tool-Using LLM Agents","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2606.17628","citing_title":"OPD-Evolver: Cultivating Holistic Agent Evolver via On-Policy Distillation","ref_index":93,"is_internal_anchor":false},{"citing_arxiv_id":"2606.17041","citing_title":"MetaSyn: A Benchmark for LLM Agents on Meta-Analysis Articles from Nature Portfolio","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2606.15932","citing_title":"Beyond NL2Code: A Structured Survey of Multimodal Code Intelligence","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2606.11176","citing_title":"Data Journalist Agent: Transforming Data into Verifiable Multimodal Stories","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2606.09550","citing_title":"InquiTree: Evaluating AI Agents in the Scientific Inquiry Loop with Paper-Derived Research Trees","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2606.09774","citing_title":"Auto-Configuring Scientific Simulators with Lightweight Coding-Agent Adapters","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2606.17041","citing_title":"MetaSyn: A Benchmark for LLM Agents on Meta-Analysis Articles from Nature Portfolio","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17373","citing_title":"FML-bench: A Controlled Study of AI Research Agent Strategies from the Perspective of Search Dynamics","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2606.17041","citing_title":"MetaSyn: A Benchmark for LLM Agents on Meta-Analysis Articles from Nature Portfolio","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2606.09774","citing_title":"Auto-Configuring Scientific Simulators with Lightweight Coding-Agent Adapters","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2606.02258","citing_title":"Matter to Mechanism: A Benchmark for AI Co-Scientists in Materials and Battery Research","ref_index":43,"is_internal_anchor":false},{"citing_arxiv_id":"2606.11926","citing_title":"Toward Generalist Autonomous Research via Hypothesis-Tree Refinement","ref_index":118,"is_internal_anchor":false},{"citing_arxiv_id":"2502.02871","citing_title":"Position: Multimodal Large Language Models Can Significantly Advance Scientific Reasoning","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2511.21016","citing_title":"Gated KalmaNet: A Fading Memory Layer Through Test-Time Ridge Regression","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2605.04375","citing_title":"Experiment-as-Code Labs: A Declarative Stack for AI-Driven Scientific Discovery","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2605.16616","citing_title":"MLReplicate: Benchmarking Autonomous Research Systems for Machine Learning Reproducibility","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17373","citing_title":"FML-bench: A Controlled Study of AI Research Agent Strategies from the Perspective of Search Dynamics","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18630","citing_title":"SCICONVBENCH: Benchmarking LLMs on Multi-Turn Clarification for Task Formulation in Computational Science","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2605.19156","citing_title":"How Far Are We From True Auto-Research?","ref_index":22,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/YZV72VZW4RF3CYA32WT76S4HT7","json":"https://pith.science/pith/YZV72VZW4RF3CYA32WT76S4HT7.json","graph_json":"https://pith.science/api/pith-number/YZV72VZW4RF3CYA32WT76S4HT7/graph.json","events_json":"https://pith.science/api/pith-number/YZV72VZW4RF3CYA32WT76S4HT7/events.json","paper":"https://pith.science/paper/YZV72VZW"},"agent_actions":{"view_html":"https://pith.science/pith/YZV72VZW4RF3CYA32WT76S4HT7","download_json":"https://pith.science/pith/YZV72VZW4RF3CYA32WT76S4HT7.json","view_paper":"https://pith.science/paper/YZV72VZW","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.05080&json=true","fetch_graph":"https://pith.science/api/pith-number/YZV72VZW4RF3CYA32WT76S4HT7/graph.json","fetch_events":"https://pith.science/api/pith-number/YZV72VZW4RF3CYA32WT76S4HT7/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/YZV72VZW4RF3CYA32WT76S4HT7/action/timestamp_anchor","attest_storage":"https://pith.science/pith/YZV72VZW4RF3CYA32WT76S4HT7/action/storage_attestation","attest_author":"https://pith.science/pith/YZV72VZW4RF3CYA32WT76S4HT7/action/author_attestation","sign_citation":"https://pith.science/pith/YZV72VZW4RF3CYA32WT76S4HT7/action/citation_signature","submit_replication":"https://pith.science/pith/YZV72VZW4RF3CYA32WT76S4HT7/action/replication_record"}},"created_at":"2026-07-05T10:41:49.045954+00:00","updated_at":"2026-07-05T10:41:49.045954+00:00"}