{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:SIFPEXAI2POCMBO7AWQ62IGUZX","short_pith_number":"pith:SIFPEXAI","schema_version":"1.0","canonical_sha256":"920af25c08d3dc2605df05a1ed20d4cdfe89471f5ef9f5311b86875680fa7b25","source":{"kind":"arxiv","id":"2502.13897","version":1},"attestation_state":"computed","paper":{"title":"DataSciBench: An LLM Agent Benchmark for Data Science","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Dan Zhang, Fengzu Li, Jie Tang, Lekang Yang, Min Cai, Sining Zhoubian, Tianjiao Dong, Wei Wang, Yisong Yue, Ziniu Hu","submitted_at":"2025-02-19T17:31:51Z","abstract_excerpt":"This paper presents DataSciBench, a comprehensive benchmark for evaluating Large Language Model (LLM) capabilities in data science. Recent related benchmarks have primarily focused on single tasks, easily obtainable ground truth, and straightforward evaluation metrics, which limits the scope of tasks that can be evaluated. In contrast, DataSciBench is constructed based on a more comprehensive and curated collection of natural and challenging prompts for uncertain ground truth and evaluation metrics. We develop a semi-automated pipeline for generating ground truth (GT) and validating evaluation"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.13897","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-02-19T17:31:51Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"aeb8b39e0ac06870fe7646ab41d020de80a5427a3be24be66e89eaa6308c0523","abstract_canon_sha256":"9a32677cb16a9c2bc04165deeed14b6ae6ad2f7b55c7c44faa65be632e030df1"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:17:02.304910Z","signature_b64":"VQNM5L/L7o05iwi/KXpyNQmMbXiXJC2yz3v/IJTi6kEF7sQq65NO7428XXUKglkS0ExyG8EwCrFvPLkRNuDcCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"920af25c08d3dc2605df05a1ed20d4cdfe89471f5ef9f5311b86875680fa7b25","last_reissued_at":"2026-07-05T10:17:02.304507Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:17:02.304507Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"DataSciBench: An LLM Agent Benchmark for Data Science","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Dan Zhang, Fengzu Li, Jie Tang, Lekang Yang, Min Cai, Sining Zhoubian, Tianjiao Dong, Wei Wang, Yisong Yue, Ziniu Hu","submitted_at":"2025-02-19T17:31:51Z","abstract_excerpt":"This paper presents DataSciBench, a comprehensive benchmark for evaluating Large Language Model (LLM) capabilities in data science. Recent related benchmarks have primarily focused on single tasks, easily obtainable ground truth, and straightforward evaluation metrics, which limits the scope of tasks that can be evaluated. In contrast, DataSciBench is constructed based on a more comprehensive and curated collection of natural and challenging prompts for uncertain ground truth and evaluation metrics. We develop a semi-automated pipeline for generating ground truth (GT) and validating evaluation"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.13897","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.13897/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.13897","created_at":"2026-07-05T10:17:02.304554+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.13897v1","created_at":"2026-07-05T10:17:02.304554+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.13897","created_at":"2026-07-05T10:17:02.304554+00:00"},{"alias_kind":"pith_short_12","alias_value":"SIFPEXAI2POC","created_at":"2026-07-05T10:17:02.304554+00:00"},{"alias_kind":"pith_short_16","alias_value":"SIFPEXAI2POCMBO7","created_at":"2026-07-05T10:17:02.304554+00:00"},{"alias_kind":"pith_short_8","alias_value":"SIFPEXAI","created_at":"2026-07-05T10:17:02.304554+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":11,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.07504","citing_title":"Do LLM-Generated Skills Make Better AI Data Scientists? A Component Ablation Across Data-Science Workflows","ref_index":17,"is_internal_anchor":true},{"citing_arxiv_id":"2606.16000","citing_title":"GRACE-DS: a Guarded Reward-guided Agent Correction Environment in Data Science","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2607.01647","citing_title":"AgenticDataBench: A Comprehensive Benchmark for Data Agents","ref_index":61,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12376","citing_title":"ProfiliTable: Profiling-Driven Tabular Data Processing via Agentic Workflows","ref_index":46,"is_internal_anchor":false},{"citing_arxiv_id":"2606.00051","citing_title":"Business Utility of Large Language Models as Exploratory Data Analysis Agents","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2605.21338","citing_title":"Text Analytics Evaluation Framework: A Case Study on LLMs and Social Media","ref_index":100,"is_internal_anchor":false},{"citing_arxiv_id":"2506.03610","citing_title":"Orak: A Foundational Benchmark for Training and Evaluating LLM Agents on Diverse Video Games","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2506.22598","citing_title":"RExBench: Can coding agents autonomously implement AI research extensions?","ref_index":54,"is_internal_anchor":false},{"citing_arxiv_id":"2604.03242","citing_title":"DRAFT: Task Decoupled Latent Reasoning for Agent Safety","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2504.19678","citing_title":"From LLM Reasoning to Autonomous AI Agents: A Comprehensive Review","ref_index":123,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12376","citing_title":"ProfiliTable: Profiling-Driven Tabular Data Processing via Agentic Workflows","ref_index":42,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/SIFPEXAI2POCMBO7AWQ62IGUZX","json":"https://pith.science/pith/SIFPEXAI2POCMBO7AWQ62IGUZX.json","graph_json":"https://pith.science/api/pith-number/SIFPEXAI2POCMBO7AWQ62IGUZX/graph.json","events_json":"https://pith.science/api/pith-number/SIFPEXAI2POCMBO7AWQ62IGUZX/events.json","paper":"https://pith.science/paper/SIFPEXAI"},"agent_actions":{"view_html":"https://pith.science/pith/SIFPEXAI2POCMBO7AWQ62IGUZX","download_json":"https://pith.science/pith/SIFPEXAI2POCMBO7AWQ62IGUZX.json","view_paper":"https://pith.science/paper/SIFPEXAI","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.13897&json=true","fetch_graph":"https://pith.science/api/pith-number/SIFPEXAI2POCMBO7AWQ62IGUZX/graph.json","fetch_events":"https://pith.science/api/pith-number/SIFPEXAI2POCMBO7AWQ62IGUZX/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/SIFPEXAI2POCMBO7AWQ62IGUZX/action/timestamp_anchor","attest_storage":"https://pith.science/pith/SIFPEXAI2POCMBO7AWQ62IGUZX/action/storage_attestation","attest_author":"https://pith.science/pith/SIFPEXAI2POCMBO7AWQ62IGUZX/action/author_attestation","sign_citation":"https://pith.science/pith/SIFPEXAI2POCMBO7AWQ62IGUZX/action/citation_signature","submit_replication":"https://pith.science/pith/SIFPEXAI2POCMBO7AWQ62IGUZX/action/replication_record"}},"created_at":"2026-07-05T10:17:02.304554+00:00","updated_at":"2026-07-05T10:17:02.304554+00:00"}