{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:J5UITM2Z74MH4C4UWY5NJVYIRC","short_pith_number":"pith:J5UITM2Z","schema_version":"1.0","canonical_sha256":"4f6889b359ff187e0b94b63ad4d70888bfe5612b9cd5f68fd9d12e3dec3217ba","source":{"kind":"arxiv","id":"2312.04528","version":2},"attestation_state":"computed","paper":{"title":"Using Large Language Models for Hyperparameter Optimization","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Jimmy Ba, Jonathan Lorraine, Juhan Bae, Michael R. Zhang, Nishkrit Desai","submitted_at":"2023-12-07T18:46:50Z","abstract_excerpt":"This paper explores the use of foundational large language models (LLMs) in hyperparameter optimization (HPO). Hyperparameters are critical in determining the effectiveness of machine learning models, yet their optimization often relies on manual approaches in limited-budget settings. By prompting LLMs with dataset and model descriptions, we develop a methodology where LLMs suggest hyperparameter configurations, which are iteratively refined based on model performance. Our empirical evaluations on standard benchmarks reveal that within constrained search budgets, LLMs can match or outperform t"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2312.04528","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2023-12-07T18:46:50Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"73185a0ae47b1c4a9a5b219aa97cf3ca691204967a75c135aced9525072abfcc","abstract_canon_sha256":"5894251f207aee1388655f00abc96f19d4f1e3d0ef770d4d26161f2929395b1e"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:33:27.271003Z","signature_b64":"OKNYKF3q8G4wGTRuoyr8IL3v7MouKImMxy5stpQkkPH7beG9THzLQAexpRPt1eLIC3NL6vL4XNRp9ZB6qPyRCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"4f6889b359ff187e0b94b63ad4d70888bfe5612b9cd5f68fd9d12e3dec3217ba","last_reissued_at":"2026-07-05T09:33:27.270515Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:33:27.270515Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Using Large Language Models for Hyperparameter Optimization","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Jimmy Ba, Jonathan Lorraine, Juhan Bae, Michael R. Zhang, Nishkrit Desai","submitted_at":"2023-12-07T18:46:50Z","abstract_excerpt":"This paper explores the use of foundational large language models (LLMs) in hyperparameter optimization (HPO). Hyperparameters are critical in determining the effectiveness of machine learning models, yet their optimization often relies on manual approaches in limited-budget settings. By prompting LLMs with dataset and model descriptions, we develop a methodology where LLMs suggest hyperparameter configurations, which are iteratively refined based on model performance. Our empirical evaluations on standard benchmarks reveal that within constrained search budgets, LLMs can match or outperform t"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2312.04528","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2312.04528/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2312.04528","created_at":"2026-07-05T09:33:27.270596+00:00"},{"alias_kind":"arxiv_version","alias_value":"2312.04528v2","created_at":"2026-07-05T09:33:27.270596+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2312.04528","created_at":"2026-07-05T09:33:27.270596+00:00"},{"alias_kind":"pith_short_12","alias_value":"J5UITM2Z74MH","created_at":"2026-07-05T09:33:27.270596+00:00"},{"alias_kind":"pith_short_16","alias_value":"J5UITM2Z74MH4C4U","created_at":"2026-07-05T09:33:27.270596+00:00"},{"alias_kind":"pith_short_8","alias_value":"J5UITM2Z","created_at":"2026-07-05T09:33:27.270596+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.25207","citing_title":"ASAP: Agent-System Co-Design for Wall-Clock-Centered Auto HPO Research for ML Experiments","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2606.23299","citing_title":"GRIMIP: A General Framework for Instance-Specific Configuration of MIP Solvers Using LLMs","ref_index":69,"is_internal_anchor":false},{"citing_arxiv_id":"2606.24901","citing_title":"LLM Evolution as an Industry-Scale Ecosystem: A Lifecycle Perspective on Continual Learning","ref_index":137,"is_internal_anchor":false},{"citing_arxiv_id":"2606.05489","citing_title":"LLM-Guided ANN Index Optimization for Human-Object Interaction Retrieval","ref_index":37,"is_internal_anchor":false},{"citing_arxiv_id":"2606.29871","citing_title":"AI Training Manager: Bounded Closed-Loop Control of Adaptive Training Recipes","ref_index":19,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/J5UITM2Z74MH4C4UWY5NJVYIRC","json":"https://pith.science/pith/J5UITM2Z74MH4C4UWY5NJVYIRC.json","graph_json":"https://pith.science/api/pith-number/J5UITM2Z74MH4C4UWY5NJVYIRC/graph.json","events_json":"https://pith.science/api/pith-number/J5UITM2Z74MH4C4UWY5NJVYIRC/events.json","paper":"https://pith.science/paper/J5UITM2Z"},"agent_actions":{"view_html":"https://pith.science/pith/J5UITM2Z74MH4C4UWY5NJVYIRC","download_json":"https://pith.science/pith/J5UITM2Z74MH4C4UWY5NJVYIRC.json","view_paper":"https://pith.science/paper/J5UITM2Z","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2312.04528&json=true","fetch_graph":"https://pith.science/api/pith-number/J5UITM2Z74MH4C4UWY5NJVYIRC/graph.json","fetch_events":"https://pith.science/api/pith-number/J5UITM2Z74MH4C4UWY5NJVYIRC/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/J5UITM2Z74MH4C4UWY5NJVYIRC/action/timestamp_anchor","attest_storage":"https://pith.science/pith/J5UITM2Z74MH4C4UWY5NJVYIRC/action/storage_attestation","attest_author":"https://pith.science/pith/J5UITM2Z74MH4C4UWY5NJVYIRC/action/author_attestation","sign_citation":"https://pith.science/pith/J5UITM2Z74MH4C4UWY5NJVYIRC/action/citation_signature","submit_replication":"https://pith.science/pith/J5UITM2Z74MH4C4UWY5NJVYIRC/action/replication_record"}},"created_at":"2026-07-05T09:33:27.270596+00:00","updated_at":"2026-07-05T09:33:27.270596+00:00"}