{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:VPFP4QF4XW7RPODNKVO7TNGZAU","short_pith_number":"pith:VPFP4QF4","schema_version":"1.0","canonical_sha256":"abcafe40bcbdbf17b86d555df9b4d9053614f1fadda8e6c49333c662955a59df","source":{"kind":"arxiv","id":"2402.17453","version":5},"attestation_state":"computed","paper":{"title":"DS-Agent: Automated Data Science by Empowering Large Language Models with Case-Based Reasoning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Cheng Deng, Hechang Chen, Jun Wang, Siyuan Guo, Yi Chang, Ying Wen","submitted_at":"2024-02-27T12:26:07Z","abstract_excerpt":"In this work, we investigate the potential of large language models (LLMs) based agents to automate data science tasks, with the goal of comprehending task requirements, then building and training the best-fit machine learning models. Despite their widespread success, existing LLM agents are hindered by generating unreasonable experiment plans within this scenario. To this end, we present DS-Agent, a novel automatic framework that harnesses LLM agent and case-based reasoning (CBR). In the development stage, DS-Agent follows the CBR framework to structure an automatic iteration pipeline, which "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2402.17453","kind":"arxiv","version":5},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-02-27T12:26:07Z","cross_cats_sorted":[],"title_canon_sha256":"680bc9eb0f8924a29a5d4c8993cd6df2a3e5fd8460c25486a8109ee73c679197","abstract_canon_sha256":"bfae4087148714909d01e983c3b1f8ba2a3ebf407b01dde6cf6aa1605f4770ac"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:24:02.870060Z","signature_b64":"8SmHWuXwSj//5ZTWn9w1XWDPDaZaLgGsGY3sp63N3YLMJVaOnE+jEhaiWbt4DrULW0GA9SSjG0HaK9xKxBn7BA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"abcafe40bcbdbf17b86d555df9b4d9053614f1fadda8e6c49333c662955a59df","last_reissued_at":"2026-07-05T08:24:02.869645Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:24:02.869645Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"DS-Agent: Automated Data Science by Empowering Large Language Models with Case-Based Reasoning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Cheng Deng, Hechang Chen, Jun Wang, Siyuan Guo, Yi Chang, Ying Wen","submitted_at":"2024-02-27T12:26:07Z","abstract_excerpt":"In this work, we investigate the potential of large language models (LLMs) based agents to automate data science tasks, with the goal of comprehending task requirements, then building and training the best-fit machine learning models. Despite their widespread success, existing LLM agents are hindered by generating unreasonable experiment plans within this scenario. To this end, we present DS-Agent, a novel automatic framework that harnesses LLM agent and case-based reasoning (CBR). In the development stage, DS-Agent follows the CBR framework to structure an automatic iteration pipeline, which "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2402.17453","kind":"arxiv","version":5},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2402.17453/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2402.17453","created_at":"2026-07-05T08:24:02.869701+00:00"},{"alias_kind":"arxiv_version","alias_value":"2402.17453v5","created_at":"2026-07-05T08:24:02.869701+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2402.17453","created_at":"2026-07-05T08:24:02.869701+00:00"},{"alias_kind":"pith_short_12","alias_value":"VPFP4QF4XW7R","created_at":"2026-07-05T08:24:02.869701+00:00"},{"alias_kind":"pith_short_16","alias_value":"VPFP4QF4XW7RPODN","created_at":"2026-07-05T08:24:02.869701+00:00"},{"alias_kind":"pith_short_8","alias_value":"VPFP4QF4","created_at":"2026-07-05T08:24:02.869701+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":18,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.25996","citing_title":"Autodata: An agentic data scientist to create high quality synthetic data","ref_index":95,"is_internal_anchor":false},{"citing_arxiv_id":"2606.25207","citing_title":"ASAP: Agent-System Co-Design for Wall-Clock-Centered Auto HPO Research for ML Experiments","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2606.25996","citing_title":"Autodata: An agentic data scientist to create high quality synthetic data","ref_index":95,"is_internal_anchor":false},{"citing_arxiv_id":"2606.17915","citing_title":"Trustworthy Self-Composable Big-Data-as-a-Service: An LLM-Orchestrated Multi-Agent Framework for Automated Data Engineering, AutoML, MLOps Deployment, and Drift-Aware Lifecycle Optimization","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2606.09365","citing_title":"Experience Makes Skillful: Enabling Generalizable Medical Agent Reasoning via Self-Evolving Skill Memory","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2606.05250","citing_title":"Towards Persistent Case-Based Memory for Autonomous Data Science: A CBR-Augmented R&D-Agent with a Locally Deployable Small Language Model","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07358","citing_title":"A Comprehensive Survey on Agent Skills: Taxonomy, Techniques, and Applications","ref_index":49,"is_internal_anchor":false},{"citing_arxiv_id":"2503.21460","citing_title":"Large Language Model Agent: A Survey on Methodology, Applications and Challenges","ref_index":116,"is_internal_anchor":false},{"citing_arxiv_id":"2509.23986","citing_title":"TusoAI: Agentic Optimization for Scientific Methods","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07358","citing_title":"A Comprehensive Survey on Agent Skills: Taxonomy, Techniques, and Applications","ref_index":51,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18661","citing_title":"AI for Auto-Research: Roadmap & User Guide","ref_index":58,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10362","citing_title":"CellDX AI Autopilot: Agent-Guided Training and Deployment of Pathology Classifiers","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2604.14655","citing_title":"AgentGA: Evolving Code Solutions in Agent-Seed Space","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2605.03808","citing_title":"Agentic-imodels: Evolving agentic interpretability tools via autoresearch","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2604.12258","citing_title":"Coding-Free and Privacy-Preserving Agentic Framework for Data-Driven Clinical Research","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2604.08491","citing_title":"Figures as Interfaces: Toward LLM-Native Artifacts for Scientific Discovery","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07358","citing_title":"A Comprehensive Survey on Agent Skills: Taxonomy, Techniques, and Applications","ref_index":51,"is_internal_anchor":false},{"citing_arxiv_id":"2604.14655","citing_title":"AgentGA: Evolving Code Solutions in Agent-Seed Space","ref_index":9,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/VPFP4QF4XW7RPODNKVO7TNGZAU","json":"https://pith.science/pith/VPFP4QF4XW7RPODNKVO7TNGZAU.json","graph_json":"https://pith.science/api/pith-number/VPFP4QF4XW7RPODNKVO7TNGZAU/graph.json","events_json":"https://pith.science/api/pith-number/VPFP4QF4XW7RPODNKVO7TNGZAU/events.json","paper":"https://pith.science/paper/VPFP4QF4"},"agent_actions":{"view_html":"https://pith.science/pith/VPFP4QF4XW7RPODNKVO7TNGZAU","download_json":"https://pith.science/pith/VPFP4QF4XW7RPODNKVO7TNGZAU.json","view_paper":"https://pith.science/paper/VPFP4QF4","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2402.17453&json=true","fetch_graph":"https://pith.science/api/pith-number/VPFP4QF4XW7RPODNKVO7TNGZAU/graph.json","fetch_events":"https://pith.science/api/pith-number/VPFP4QF4XW7RPODNKVO7TNGZAU/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/VPFP4QF4XW7RPODNKVO7TNGZAU/action/timestamp_anchor","attest_storage":"https://pith.science/pith/VPFP4QF4XW7RPODNKVO7TNGZAU/action/storage_attestation","attest_author":"https://pith.science/pith/VPFP4QF4XW7RPODNKVO7TNGZAU/action/author_attestation","sign_citation":"https://pith.science/pith/VPFP4QF4XW7RPODNKVO7TNGZAU/action/citation_signature","submit_replication":"https://pith.science/pith/VPFP4QF4XW7RPODNKVO7TNGZAU/action/replication_record"}},"created_at":"2026-07-05T08:24:02.869701+00:00","updated_at":"2026-07-05T08:24:02.869701+00:00"}