{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:ZUEB7VGEQ4WCOIPWFLM2UMJIG7","short_pith_number":"pith:ZUEB7VGE","schema_version":"1.0","canonical_sha256":"cd081fd4c4872c2721f62ad9aa312837eacb09e97b1d86c0dcf6e6f1ab37f6e5","source":{"kind":"arxiv","id":"2411.10478","version":2},"attestation_state":"computed","paper":{"title":"Large Language Models for Constructing and Optimizing Machine Learning Workflows: A Survey","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Haoran Fan, Hengyu You, Jian Cao, Muran Yu, Shiyou Qian, Yang Gu","submitted_at":"2024-11-11T21:54:26Z","abstract_excerpt":"Building effective machine learning (ML) workflows to address complex tasks is a primary focus of the Automatic ML (AutoML) community and a critical step toward achieving artificial general intelligence (AGI). Recently, the integration of Large Language Models (LLMs) into ML workflows has shown great potential for automating and enhancing various stages of the ML pipeline. This survey provides a comprehensive and up-to-date review of recent advancements in using LLMs to construct and optimize ML workflows, focusing on key components encompassing data and feature engineering, model selection an"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2411.10478","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-11-11T21:54:26Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"4b7388f4ae5a05cbdd14c5d7489436000754d21fb9d22cd74abad736ae97381f","abstract_canon_sha256":"53525f62a1272d5cae7cbcc7a78861ef2066a76329a7a7cd3a4d35925e5a6a91"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:54:05.791408Z","signature_b64":"zNHYicRi7cxY8ebZdxoHi9UqAkpwsK9X1OE93Ua/BzcmuqoN4lm8FCurwh8hQoMKxqmAqYPzCvm2wL+gu6L1DQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"cd081fd4c4872c2721f62ad9aa312837eacb09e97b1d86c0dcf6e6f1ab37f6e5","last_reissued_at":"2026-07-05T09:54:05.790914Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:54:05.790914Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Large Language Models for Constructing and Optimizing Machine Learning Workflows: A Survey","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Haoran Fan, Hengyu You, Jian Cao, Muran Yu, Shiyou Qian, Yang Gu","submitted_at":"2024-11-11T21:54:26Z","abstract_excerpt":"Building effective machine learning (ML) workflows to address complex tasks is a primary focus of the Automatic ML (AutoML) community and a critical step toward achieving artificial general intelligence (AGI). Recently, the integration of Large Language Models (LLMs) into ML workflows has shown great potential for automating and enhancing various stages of the ML pipeline. This survey provides a comprehensive and up-to-date review of recent advancements in using LLMs to construct and optimize ML workflows, focusing on key components encompassing data and feature engineering, model selection an"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2411.10478","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2411.10478/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2411.10478","created_at":"2026-07-05T09:54:05.790972+00:00"},{"alias_kind":"arxiv_version","alias_value":"2411.10478v2","created_at":"2026-07-05T09:54:05.790972+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2411.10478","created_at":"2026-07-05T09:54:05.790972+00:00"},{"alias_kind":"pith_short_12","alias_value":"ZUEB7VGEQ4WC","created_at":"2026-07-05T09:54:05.790972+00:00"},{"alias_kind":"pith_short_16","alias_value":"ZUEB7VGEQ4WCOIPW","created_at":"2026-07-05T09:54:05.790972+00:00"},{"alias_kind":"pith_short_8","alias_value":"ZUEB7VGE","created_at":"2026-07-05T09:54:05.790972+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2604.11945","citing_title":"AutoSurrogate: An LLM-Driven Multi-Agent Framework for Autonomous Construction of Deep Learning Surrogate Models in Subsurface Flow","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2604.20261","citing_title":"Memory-Augmented LLM-based Multi-Agent System for Automated Feature Generation on Tabular Data","ref_index":61,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ZUEB7VGEQ4WCOIPWFLM2UMJIG7","json":"https://pith.science/pith/ZUEB7VGEQ4WCOIPWFLM2UMJIG7.json","graph_json":"https://pith.science/api/pith-number/ZUEB7VGEQ4WCOIPWFLM2UMJIG7/graph.json","events_json":"https://pith.science/api/pith-number/ZUEB7VGEQ4WCOIPWFLM2UMJIG7/events.json","paper":"https://pith.science/paper/ZUEB7VGE"},"agent_actions":{"view_html":"https://pith.science/pith/ZUEB7VGEQ4WCOIPWFLM2UMJIG7","download_json":"https://pith.science/pith/ZUEB7VGEQ4WCOIPWFLM2UMJIG7.json","view_paper":"https://pith.science/paper/ZUEB7VGE","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2411.10478&json=true","fetch_graph":"https://pith.science/api/pith-number/ZUEB7VGEQ4WCOIPWFLM2UMJIG7/graph.json","fetch_events":"https://pith.science/api/pith-number/ZUEB7VGEQ4WCOIPWFLM2UMJIG7/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ZUEB7VGEQ4WCOIPWFLM2UMJIG7/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ZUEB7VGEQ4WCOIPWFLM2UMJIG7/action/storage_attestation","attest_author":"https://pith.science/pith/ZUEB7VGEQ4WCOIPWFLM2UMJIG7/action/author_attestation","sign_citation":"https://pith.science/pith/ZUEB7VGEQ4WCOIPWFLM2UMJIG7/action/citation_signature","submit_replication":"https://pith.science/pith/ZUEB7VGEQ4WCOIPWFLM2UMJIG7/action/replication_record"}},"created_at":"2026-07-05T09:54:05.790972+00:00","updated_at":"2026-07-05T09:54:05.790972+00:00"}