{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:BSTYO6FTFOX7BMQYL4KSX44FRI","short_pith_number":"pith:BSTYO6FT","schema_version":"1.0","canonical_sha256":"0ca78778b32baff0b2185f152bf3858a3b05f4a0b7c9ddedcef9270997018581","source":{"kind":"arxiv","id":"2503.06101","version":2},"attestation_state":"computed","paper":{"title":"ULTHO: Ultra-Lightweight yet Efficient Hyperparameter Optimization in Deep Reinforcement Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Bo Li, Mingqi Yuan, Wenjun Zeng, Xin Jin","submitted_at":"2025-03-08T07:03:43Z","abstract_excerpt":"Hyperparameter optimization (HPO) is a billion-dollar problem in machine learning, which significantly impacts the training efficiency and model performance. However, achieving efficient and robust HPO in deep reinforcement learning (RL) is consistently challenging due to its high non-stationarity and computational cost. To tackle this problem, existing approaches attempt to adapt common HPO techniques (e.g., population-based training or Bayesian optimization) to the RL scenario. However, they remain sample-inefficient and computationally expensive, which cannot facilitate a wide range of appl"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2503.06101","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2025-03-08T07:03:43Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"0d1d0ab713455fa708bec33bbe29ecb2de391ec7c4b0ab065823b98f483b0596","abstract_canon_sha256":"58f1034182f1318c1ac3839946ef32676b6690909053298b348a21bd8eb6617b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:46:40.776905Z","signature_b64":"NyR8lmojV0QoyGDMo3X+U7VQdgf4SzO5nV4Nafe8IO0FdV9A+puAoAQzhnRdk1DDFoojM1Egq8v8LVhNGM7VAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"0ca78778b32baff0b2185f152bf3858a3b05f4a0b7c9ddedcef9270997018581","last_reissued_at":"2026-07-05T11:46:40.776416Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:46:40.776416Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"ULTHO: Ultra-Lightweight yet Efficient Hyperparameter Optimization in Deep Reinforcement Learning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Bo Li, Mingqi Yuan, Wenjun Zeng, Xin Jin","submitted_at":"2025-03-08T07:03:43Z","abstract_excerpt":"Hyperparameter optimization (HPO) is a billion-dollar problem in machine learning, which significantly impacts the training efficiency and model performance. However, achieving efficient and robust HPO in deep reinforcement learning (RL) is consistently challenging due to its high non-stationarity and computational cost. To tackle this problem, existing approaches attempt to adapt common HPO techniques (e.g., population-based training or Bayesian optimization) to the RL scenario. However, they remain sample-inefficient and computationally expensive, which cannot facilitate a wide range of appl"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2503.06101","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2503.06101/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2503.06101","created_at":"2026-07-05T11:46:40.776480+00:00"},{"alias_kind":"arxiv_version","alias_value":"2503.06101v2","created_at":"2026-07-05T11:46:40.776480+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2503.06101","created_at":"2026-07-05T11:46:40.776480+00:00"},{"alias_kind":"pith_short_12","alias_value":"BSTYO6FTFOX7","created_at":"2026-07-05T11:46:40.776480+00:00"},{"alias_kind":"pith_short_16","alias_value":"BSTYO6FTFOX7BMQY","created_at":"2026-07-05T11:46:40.776480+00:00"},{"alias_kind":"pith_short_8","alias_value":"BSTYO6FT","created_at":"2026-07-05T11:46:40.776480+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2510.03508","citing_title":"D2 Actor Critic: Diffusion Actor Meets Distributional Critic","ref_index":38,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/BSTYO6FTFOX7BMQYL4KSX44FRI","json":"https://pith.science/pith/BSTYO6FTFOX7BMQYL4KSX44FRI.json","graph_json":"https://pith.science/api/pith-number/BSTYO6FTFOX7BMQYL4KSX44FRI/graph.json","events_json":"https://pith.science/api/pith-number/BSTYO6FTFOX7BMQYL4KSX44FRI/events.json","paper":"https://pith.science/paper/BSTYO6FT"},"agent_actions":{"view_html":"https://pith.science/pith/BSTYO6FTFOX7BMQYL4KSX44FRI","download_json":"https://pith.science/pith/BSTYO6FTFOX7BMQYL4KSX44FRI.json","view_paper":"https://pith.science/paper/BSTYO6FT","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2503.06101&json=true","fetch_graph":"https://pith.science/api/pith-number/BSTYO6FTFOX7BMQYL4KSX44FRI/graph.json","fetch_events":"https://pith.science/api/pith-number/BSTYO6FTFOX7BMQYL4KSX44FRI/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/BSTYO6FTFOX7BMQYL4KSX44FRI/action/timestamp_anchor","attest_storage":"https://pith.science/pith/BSTYO6FTFOX7BMQYL4KSX44FRI/action/storage_attestation","attest_author":"https://pith.science/pith/BSTYO6FTFOX7BMQYL4KSX44FRI/action/author_attestation","sign_citation":"https://pith.science/pith/BSTYO6FTFOX7BMQYL4KSX44FRI/action/citation_signature","submit_replication":"https://pith.science/pith/BSTYO6FTFOX7BMQYL4KSX44FRI/action/replication_record"}},"created_at":"2026-07-05T11:46:40.776480+00:00","updated_at":"2026-07-05T11:46:40.776480+00:00"}