{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:G6PPFFMFVFA2WZT4GMO55X5X72","short_pith_number":"pith:G6PPFFMF","schema_version":"1.0","canonical_sha256":"379ef29585a941ab667c331ddedfb7fe8c0a72308029ad375ce32477e3e64454","source":{"kind":"arxiv","id":"2410.20424","version":3},"attestation_state":"computed","paper":{"title":"AutoKaggle: A Multi-Agent Framework for Autonomous Data Science Competitions","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.AI","authors_text":"David Ma, Ge Zhang, Jiaheng Liu, Jian Yang, Jiawei Guo, Minghao Liu, Qianbo Zang, Tuney Zheng, Wangchunshu Zhou, Wanjun Zhong, Wenhao Huang, Xinyao Niu, Yue Wang, Ziming Li","submitted_at":"2024-10-27T12:44:25Z","abstract_excerpt":"Data science tasks involving tabular data present complex challenges that require sophisticated problem-solving approaches. We propose AutoKaggle, a powerful and user-centric framework that assists data scientists in completing daily data pipelines through a collaborative multi-agent system. AutoKaggle implements an iterative development process that combines code execution, debugging, and comprehensive unit testing to ensure code correctness and logic consistency. The framework offers highly customizable workflows, allowing users to intervene at each phase, thus integrating automated intellig"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.20424","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by-sa/4.0/","primary_cat":"cs.AI","submitted_at":"2024-10-27T12:44:25Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"c70ab3c0c79467a5a21a0567961e2770a192a72913732d30c0f0cec776c47ab3","abstract_canon_sha256":"05279ef695aa06e5a50fb014b41a9dff09129702c5007a03028058d83b37f2c4"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:31:31.015598Z","signature_b64":"G+f97DHTMwsdqAqsVeet1ItAD7WBECDYxr+wJX9cUO+30g6GMMq5ffwHKJLv+Y2LNK6ZkGmEFo94922OBa3nDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"379ef29585a941ab667c331ddedfb7fe8c0a72308029ad375ce32477e3e64454","last_reissued_at":"2026-07-05T09:31:31.014894Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:31:31.014894Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"AutoKaggle: A Multi-Agent Framework for Autonomous Data Science Competitions","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.AI","authors_text":"David Ma, Ge Zhang, Jiaheng Liu, Jian Yang, Jiawei Guo, Minghao Liu, Qianbo Zang, Tuney Zheng, Wangchunshu Zhou, Wanjun Zhong, Wenhao Huang, Xinyao Niu, Yue Wang, Ziming Li","submitted_at":"2024-10-27T12:44:25Z","abstract_excerpt":"Data science tasks involving tabular data present complex challenges that require sophisticated problem-solving approaches. We propose AutoKaggle, a powerful and user-centric framework that assists data scientists in completing daily data pipelines through a collaborative multi-agent system. AutoKaggle implements an iterative development process that combines code execution, debugging, and comprehensive unit testing to ensure code correctness and logic consistency. The framework offers highly customizable workflows, allowing users to intervene at each phase, thus integrating automated intellig"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.20424","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.20424/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.20424","created_at":"2026-07-05T09:31:31.014967+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.20424v3","created_at":"2026-07-05T09:31:31.014967+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.20424","created_at":"2026-07-05T09:31:31.014967+00:00"},{"alias_kind":"pith_short_12","alias_value":"G6PPFFMFVFA2","created_at":"2026-07-05T09:31:31.014967+00:00"},{"alias_kind":"pith_short_16","alias_value":"G6PPFFMFVFA2WZT4","created_at":"2026-07-05T09:31:31.014967+00:00"},{"alias_kind":"pith_short_8","alias_value":"G6PPFFMF","created_at":"2026-07-05T09:31:31.014967+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":9,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.09004","citing_title":"LATTEArena: An Evaluation Framework for LLM-powered Tabular Feature Engineering (Extended Version)","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2606.05250","citing_title":"Towards Persistent Case-Based Memory for Autonomous Data Science: A CBR-Augmented R&D-Agent with a Locally Deployable Small Language Model","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2504.02181","citing_title":"A Survey of Scaling in Large Language Model Reasoning","ref_index":105,"is_internal_anchor":false},{"citing_arxiv_id":"2505.05583","citing_title":"KG-HTC: Integrating Knowledge Graphs into LLMs for Effective Zero-shot Hierarchical Text Classification","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2605.15871","citing_title":"Agentic Discovery of Neural Architectures: AIRA-Compose and AIRA-Design","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2508.10177","citing_title":"KompeteAI: Accelerated Autonomous Multi-Agent System for End-to-End Pipeline Generation for Machine Learning Problems","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2604.14655","citing_title":"AgentGA: Evolving Code Solutions in Agent-Seed Space","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2409.12186","citing_title":"Qwen2.5-Coder Technical Report","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2604.14655","citing_title":"AgentGA: Evolving Code Solutions in Agent-Seed Space","ref_index":15,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/G6PPFFMFVFA2WZT4GMO55X5X72","json":"https://pith.science/pith/G6PPFFMFVFA2WZT4GMO55X5X72.json","graph_json":"https://pith.science/api/pith-number/G6PPFFMFVFA2WZT4GMO55X5X72/graph.json","events_json":"https://pith.science/api/pith-number/G6PPFFMFVFA2WZT4GMO55X5X72/events.json","paper":"https://pith.science/paper/G6PPFFMF"},"agent_actions":{"view_html":"https://pith.science/pith/G6PPFFMFVFA2WZT4GMO55X5X72","download_json":"https://pith.science/pith/G6PPFFMFVFA2WZT4GMO55X5X72.json","view_paper":"https://pith.science/paper/G6PPFFMF","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.20424&json=true","fetch_graph":"https://pith.science/api/pith-number/G6PPFFMFVFA2WZT4GMO55X5X72/graph.json","fetch_events":"https://pith.science/api/pith-number/G6PPFFMFVFA2WZT4GMO55X5X72/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/G6PPFFMFVFA2WZT4GMO55X5X72/action/timestamp_anchor","attest_storage":"https://pith.science/pith/G6PPFFMFVFA2WZT4GMO55X5X72/action/storage_attestation","attest_author":"https://pith.science/pith/G6PPFFMFVFA2WZT4GMO55X5X72/action/author_attestation","sign_citation":"https://pith.science/pith/G6PPFFMFVFA2WZT4GMO55X5X72/action/citation_signature","submit_replication":"https://pith.science/pith/G6PPFFMFVFA2WZT4GMO55X5X72/action/replication_record"}},"created_at":"2026-07-05T09:31:31.014967+00:00","updated_at":"2026-07-05T09:31:31.014967+00:00"}