{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:KETHQCOFSKKXABHQJET6ERNZPF","short_pith_number":"pith:KETHQCOF","schema_version":"1.0","canonical_sha256":"51267809c592957004f04927e245b9796c1f28395a7ffed3e9e4819256636bbc","source":{"kind":"arxiv","id":"2405.03727","version":3},"attestation_state":"computed","paper":{"title":"Large Language Models Synergize with Automated Machine Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG","cs.PL"],"primary_cat":"cs.SE","authors_text":"Guoyuan Zhou, Hitoshi Iba, Jia Guo, Jialong Li, Jinglue Xu, Kenji Tei, Nagar Anthel Venkatesh Suryanarayanan, Zhen Liu","submitted_at":"2024-05-06T08:09:46Z","abstract_excerpt":"Recently, program synthesis driven by large language models (LLMs) has become increasingly popular. However, program synthesis for machine learning (ML) tasks still poses significant challenges. This paper explores a novel form of program synthesis, targeting ML programs, by combining LLMs and automated machine learning (autoML). Specifically, our goal is to fully automate the generation and optimization of the code of the entire ML workflow, from data preparation to modeling and post-processing, utilizing only textual descriptions of the ML tasks. To manage the length and diversity of ML prog"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2405.03727","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.SE","submitted_at":"2024-05-06T08:09:46Z","cross_cats_sorted":["cs.AI","cs.LG","cs.PL"],"title_canon_sha256":"31345cfdd5be24c4314bbfd0a2614d9306eea94ebc935bc805882f364f07ee3b","abstract_canon_sha256":"3fed11b4e94efb83a41ff76bc1190d726ca8ec4d1a576a296794b543aefe54c2"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:04:35.761370Z","signature_b64":"yu2DushAA7b/hC7EADRA2ML1/ITTOE4s2tiTnN1h3MHLTXvGQ3Mre1E7VFK5ivBJhK3iFZehNyg2slTyTIgIAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"51267809c592957004f04927e245b9796c1f28395a7ffed3e9e4819256636bbc","last_reissued_at":"2026-07-05T09:04:35.760900Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:04:35.760900Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Large Language Models Synergize with Automated Machine Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG","cs.PL"],"primary_cat":"cs.SE","authors_text":"Guoyuan Zhou, Hitoshi Iba, Jia Guo, Jialong Li, Jinglue Xu, Kenji Tei, Nagar Anthel Venkatesh Suryanarayanan, Zhen Liu","submitted_at":"2024-05-06T08:09:46Z","abstract_excerpt":"Recently, program synthesis driven by large language models (LLMs) has become increasingly popular. However, program synthesis for machine learning (ML) tasks still poses significant challenges. This paper explores a novel form of program synthesis, targeting ML programs, by combining LLMs and automated machine learning (autoML). Specifically, our goal is to fully automate the generation and optimization of the code of the entire ML workflow, from data preparation to modeling and post-processing, utilizing only textual descriptions of the ML tasks. To manage the length and diversity of ML prog"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2405.03727","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2405.03727/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2405.03727","created_at":"2026-07-05T09:04:35.760955+00:00"},{"alias_kind":"arxiv_version","alias_value":"2405.03727v3","created_at":"2026-07-05T09:04:35.760955+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2405.03727","created_at":"2026-07-05T09:04:35.760955+00:00"},{"alias_kind":"pith_short_12","alias_value":"KETHQCOFSKKX","created_at":"2026-07-05T09:04:35.760955+00:00"},{"alias_kind":"pith_short_16","alias_value":"KETHQCOFSKKXABHQ","created_at":"2026-07-05T09:04:35.760955+00:00"},{"alias_kind":"pith_short_8","alias_value":"KETHQCOF","created_at":"2026-07-05T09:04:35.760955+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2604.20261","citing_title":"Memory-Augmented LLM-based Multi-Agent System for Automated Feature Generation on Tabular Data","ref_index":90,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/KETHQCOFSKKXABHQJET6ERNZPF","json":"https://pith.science/pith/KETHQCOFSKKXABHQJET6ERNZPF.json","graph_json":"https://pith.science/api/pith-number/KETHQCOFSKKXABHQJET6ERNZPF/graph.json","events_json":"https://pith.science/api/pith-number/KETHQCOFSKKXABHQJET6ERNZPF/events.json","paper":"https://pith.science/paper/KETHQCOF"},"agent_actions":{"view_html":"https://pith.science/pith/KETHQCOFSKKXABHQJET6ERNZPF","download_json":"https://pith.science/pith/KETHQCOFSKKXABHQJET6ERNZPF.json","view_paper":"https://pith.science/paper/KETHQCOF","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2405.03727&json=true","fetch_graph":"https://pith.science/api/pith-number/KETHQCOFSKKXABHQJET6ERNZPF/graph.json","fetch_events":"https://pith.science/api/pith-number/KETHQCOFSKKXABHQJET6ERNZPF/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/KETHQCOFSKKXABHQJET6ERNZPF/action/timestamp_anchor","attest_storage":"https://pith.science/pith/KETHQCOFSKKXABHQJET6ERNZPF/action/storage_attestation","attest_author":"https://pith.science/pith/KETHQCOFSKKXABHQJET6ERNZPF/action/author_attestation","sign_citation":"https://pith.science/pith/KETHQCOFSKKXABHQJET6ERNZPF/action/citation_signature","submit_replication":"https://pith.science/pith/KETHQCOFSKKXABHQJET6ERNZPF/action/replication_record"}},"created_at":"2026-07-05T09:04:35.760955+00:00","updated_at":"2026-07-05T09:04:35.760955+00:00"}