{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:LH4RUVI3LO6LHQZCIPJWCMNPWO","short_pith_number":"pith:LH4RUVI3","schema_version":"1.0","canonical_sha256":"59f91a551b5bbcb3c32243d36131afb395a226d39758ea20a002a8916b363e02","source":{"kind":"arxiv","id":"2507.16280","version":1},"attestation_state":"computed","paper":{"title":"ResearcherBench: Evaluating Deep AI Research Systems on the Frontiers of Scientific Inquiry","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Lyumanshan Ye, Pengfei Liu, Pengrui Lu, Tianze Xu, Xiangkun Hu","submitted_at":"2025-07-22T06:51:26Z","abstract_excerpt":"The emergence of deep research systems presents significant capabilities in problem-solving, extending from basic queries to sophisticated research tasks. However, existing benchmarks primarily evaluate these systems as agents for web retrieval and report generation, overlooking their potential to discover novel insights on the frontiers of scientific research. To address this gap, we introduce ResearcherBench, the first benchmark focused on evaluating the capabilities of these advanced, agentic systems - which we refer to as Deep AI Research Systems (DARS) - on frontier AI scientific question"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2507.16280","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2025-07-22T06:51:26Z","cross_cats_sorted":[],"title_canon_sha256":"fa0ef4e62a6f1cdf210328e5789886932e95ec9a5b196ffc1b9e4111f418fbd9","abstract_canon_sha256":"ddb6c76c2b244b288cecd14242b36ae2569961e35d353de9a3b4acca44a19794"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:41:06.851557Z","signature_b64":"nxQCnHbZp1w3JxUQphuS4Kam+OBw6bxu6MiiOPJWp5jPvcpVvhlbxJjQ65ZEo8BdiU+VwuPm/VjasJdgqfekBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"59f91a551b5bbcb3c32243d36131afb395a226d39758ea20a002a8916b363e02","last_reissued_at":"2026-07-05T11:41:06.851119Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:41:06.851119Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"ResearcherBench: Evaluating Deep AI Research Systems on the Frontiers of Scientific Inquiry","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Lyumanshan Ye, Pengfei Liu, Pengrui Lu, Tianze Xu, Xiangkun Hu","submitted_at":"2025-07-22T06:51:26Z","abstract_excerpt":"The emergence of deep research systems presents significant capabilities in problem-solving, extending from basic queries to sophisticated research tasks. However, existing benchmarks primarily evaluate these systems as agents for web retrieval and report generation, overlooking their potential to discover novel insights on the frontiers of scientific research. To address this gap, we introduce ResearcherBench, the first benchmark focused on evaluating the capabilities of these advanced, agentic systems - which we refer to as Deep AI Research Systems (DARS) - on frontier AI scientific question"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2507.16280","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2507.16280/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2507.16280","created_at":"2026-07-05T11:41:06.851174+00:00"},{"alias_kind":"arxiv_version","alias_value":"2507.16280v1","created_at":"2026-07-05T11:41:06.851174+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2507.16280","created_at":"2026-07-05T11:41:06.851174+00:00"},{"alias_kind":"pith_short_12","alias_value":"LH4RUVI3LO6L","created_at":"2026-07-05T11:41:06.851174+00:00"},{"alias_kind":"pith_short_16","alias_value":"LH4RUVI3LO6LHQZC","created_at":"2026-07-05T11:41:06.851174+00:00"},{"alias_kind":"pith_short_8","alias_value":"LH4RUVI3","created_at":"2026-07-05T11:41:06.851174+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":8,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.11337","citing_title":"Can AI Agents Synthesize Scientific Conclusions?","ref_index":133,"is_internal_anchor":false},{"citing_arxiv_id":"2605.26340","citing_title":"ScientistOne: Towards Human-Level Autonomous Research via Chain-of-Evidence","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2605.23204","citing_title":"AutoResearch AI: Towards AI-Powered Research Automation for Scientific Discovery","ref_index":57,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17554","citing_title":"Evaluating Deep Research Agents on Expert Consulting Work: A Benchmark with Verifiers, Rubrics, and Cognitive Traps","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17324","citing_title":"ASPI: Seeking Ambiguity Clarification Amplifies Prompt Injection Vulnerability in LLM Agents","ref_index":85,"is_internal_anchor":false},{"citing_arxiv_id":"2601.11044","citing_title":"AgencyBench: Benchmarking the Frontiers of Autonomous Agents in 1M-Token Real-World Contexts","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10530","citing_title":"Personalized Deep Research: A User-Centric Framework, Dataset, and Hybrid Evaluation for Knowledge Discovery","ref_index":45,"is_internal_anchor":false},{"citing_arxiv_id":"2604.10718","citing_title":"SciPredict: Can LLMs Predict the Outcomes of Scientific Experiments in Natural Sciences?","ref_index":45,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/LH4RUVI3LO6LHQZCIPJWCMNPWO","json":"https://pith.science/pith/LH4RUVI3LO6LHQZCIPJWCMNPWO.json","graph_json":"https://pith.science/api/pith-number/LH4RUVI3LO6LHQZCIPJWCMNPWO/graph.json","events_json":"https://pith.science/api/pith-number/LH4RUVI3LO6LHQZCIPJWCMNPWO/events.json","paper":"https://pith.science/paper/LH4RUVI3"},"agent_actions":{"view_html":"https://pith.science/pith/LH4RUVI3LO6LHQZCIPJWCMNPWO","download_json":"https://pith.science/pith/LH4RUVI3LO6LHQZCIPJWCMNPWO.json","view_paper":"https://pith.science/paper/LH4RUVI3","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2507.16280&json=true","fetch_graph":"https://pith.science/api/pith-number/LH4RUVI3LO6LHQZCIPJWCMNPWO/graph.json","fetch_events":"https://pith.science/api/pith-number/LH4RUVI3LO6LHQZCIPJWCMNPWO/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/LH4RUVI3LO6LHQZCIPJWCMNPWO/action/timestamp_anchor","attest_storage":"https://pith.science/pith/LH4RUVI3LO6LHQZCIPJWCMNPWO/action/storage_attestation","attest_author":"https://pith.science/pith/LH4RUVI3LO6LHQZCIPJWCMNPWO/action/author_attestation","sign_citation":"https://pith.science/pith/LH4RUVI3LO6LHQZCIPJWCMNPWO/action/citation_signature","submit_replication":"https://pith.science/pith/LH4RUVI3LO6LHQZCIPJWCMNPWO/action/replication_record"}},"created_at":"2026-07-05T11:41:06.851174+00:00","updated_at":"2026-07-05T11:41:06.851174+00:00"}