{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:5S7QIIHZWBMSOHEHDFBBSVTZYD","short_pith_number":"pith:5S7QIIHZ","schema_version":"1.0","canonical_sha256":"ecbf0420f9b059271c871942195679c0c94e875afb4da22c6534cf2ca5856aa9","source":{"kind":"arxiv","id":"2409.07703","version":3},"attestation_state":"computed","paper":{"title":"DSBench: How Far Are Data Science Agents from Becoming Data Science Experts?","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.AI","authors_text":"Dong Yu, Hongming Zhang, Kaixin Ma, Liqiang Jing, Wenhao Yu, Wenlin Yao, Xiaoyang Wang, Xinya Du, Zhehui Huang","submitted_at":"2024-09-12T02:08:00Z","abstract_excerpt":"Large Language Models (LLMs) and Large Vision-Language Models (LVLMs) have demonstrated impressive language/vision reasoning abilities, igniting the recent trend of building agents for targeted applications such as shopping assistants or AI software engineers. Recently, many data science benchmarks have been proposed to investigate their performance in the data science domain. However, existing data science benchmarks still fall short when compared to real-world data science applications due to their simplified settings. To bridge this gap, we introduce DSBench, a comprehensive benchmark desig"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2409.07703","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2024-09-12T02:08:00Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"24764dfb73eb2275270ab1e671fff6f266b0caded3c1d95a53fe9ee68027106f","abstract_canon_sha256":"65a9b710d43d2dc0fd71cc3c580facf85019445ae76d997ee21dd69a3af21630"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:47:30.800541Z","signature_b64":"ol8IBgEK0teigQavwgsqTpGOlebJlPs9h1VPVARLiiXsTNntdKT+tDF+ochBK3VJpg91ynInHiM1iPLbVvvNDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ecbf0420f9b059271c871942195679c0c94e875afb4da22c6534cf2ca5856aa9","last_reissued_at":"2026-07-05T10:47:30.800062Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:47:30.800062Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"DSBench: How Far Are Data Science Agents from Becoming Data Science Experts?","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.AI","authors_text":"Dong Yu, Hongming Zhang, Kaixin Ma, Liqiang Jing, Wenhao Yu, Wenlin Yao, Xiaoyang Wang, Xinya Du, Zhehui Huang","submitted_at":"2024-09-12T02:08:00Z","abstract_excerpt":"Large Language Models (LLMs) and Large Vision-Language Models (LVLMs) have demonstrated impressive language/vision reasoning abilities, igniting the recent trend of building agents for targeted applications such as shopping assistants or AI software engineers. Recently, many data science benchmarks have been proposed to investigate their performance in the data science domain. However, existing data science benchmarks still fall short when compared to real-world data science applications due to their simplified settings. To bridge this gap, we introduce DSBench, a comprehensive benchmark desig"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2409.07703","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2409.07703/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2409.07703","created_at":"2026-07-05T10:47:30.800118+00:00"},{"alias_kind":"arxiv_version","alias_value":"2409.07703v3","created_at":"2026-07-05T10:47:30.800118+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2409.07703","created_at":"2026-07-05T10:47:30.800118+00:00"},{"alias_kind":"pith_short_12","alias_value":"5S7QIIHZWBMS","created_at":"2026-07-05T10:47:30.800118+00:00"},{"alias_kind":"pith_short_16","alias_value":"5S7QIIHZWBMSOHEH","created_at":"2026-07-05T10:47:30.800118+00:00"},{"alias_kind":"pith_short_8","alias_value":"5S7QIIHZ","created_at":"2026-07-05T10:47:30.800118+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":9,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.11176","citing_title":"Data Journalist Agent: Transforming Data into Verifiable Multimodal Stories","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2606.00051","citing_title":"Business Utility of Large Language Models as Exploratory Data Analysis Agents","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17373","citing_title":"FML-bench: A Controlled Study of AI Research Agent Strategies from the Perspective of Search Dynamics","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2410.07095","citing_title":"MLE-bench: Evaluating Machine Learning Agents on Machine Learning Engineering","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17373","citing_title":"FML-bench: A Controlled Study of AI Research Agent Strategies from the Perspective of Search Dynamics","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2506.22598","citing_title":"RExBench: Can coding agents autonomously implement AI research extensions?","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2508.10177","citing_title":"KompeteAI: Accelerated Autonomous Multi-Agent System for End-to-End Pipeline Generation for Machine Learning Problems","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2510.05307","citing_title":"When Should Users Check? Modeling Confirmation Frequency inMulti-Step Agentic AI Tasks","ref_index":36,"is_internal_anchor":false},{"citing_arxiv_id":"2604.09791","citing_title":"Pioneer Agent: Continual Improvement of Small Language Models in Production","ref_index":45,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/5S7QIIHZWBMSOHEHDFBBSVTZYD","json":"https://pith.science/pith/5S7QIIHZWBMSOHEHDFBBSVTZYD.json","graph_json":"https://pith.science/api/pith-number/5S7QIIHZWBMSOHEHDFBBSVTZYD/graph.json","events_json":"https://pith.science/api/pith-number/5S7QIIHZWBMSOHEHDFBBSVTZYD/events.json","paper":"https://pith.science/paper/5S7QIIHZ"},"agent_actions":{"view_html":"https://pith.science/pith/5S7QIIHZWBMSOHEHDFBBSVTZYD","download_json":"https://pith.science/pith/5S7QIIHZWBMSOHEHDFBBSVTZYD.json","view_paper":"https://pith.science/paper/5S7QIIHZ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2409.07703&json=true","fetch_graph":"https://pith.science/api/pith-number/5S7QIIHZWBMSOHEHDFBBSVTZYD/graph.json","fetch_events":"https://pith.science/api/pith-number/5S7QIIHZWBMSOHEHDFBBSVTZYD/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/5S7QIIHZWBMSOHEHDFBBSVTZYD/action/timestamp_anchor","attest_storage":"https://pith.science/pith/5S7QIIHZWBMSOHEHDFBBSVTZYD/action/storage_attestation","attest_author":"https://pith.science/pith/5S7QIIHZWBMSOHEHDFBBSVTZYD/action/author_attestation","sign_citation":"https://pith.science/pith/5S7QIIHZWBMSOHEHDFBBSVTZYD/action/citation_signature","submit_replication":"https://pith.science/pith/5S7QIIHZWBMSOHEHDFBBSVTZYD/action/replication_record"}},"created_at":"2026-07-05T10:47:30.800118+00:00","updated_at":"2026-07-05T10:47:30.800118+00:00"}