{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:ECKXJIATXNMVDBWLZ2ARZZFWGH","short_pith_number":"pith:ECKXJIAT","schema_version":"1.0","canonical_sha256":"209574a013bb595186cbce811ce4b631dab8692551e921ef0e0367c11db93283","source":{"kind":"arxiv","id":"2405.16195","version":3},"attestation_state":"computed","paper":{"title":"Adaptive $Q$-Network: On-the-fly Target Selection for Deep Reinforcement Learning","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Boris Belousov, Carlo D'Eramo, Fabian Wahren, Jan Peters, Th\\'eo Vincent","submitted_at":"2024-05-25T11:57:43Z","abstract_excerpt":"Deep Reinforcement Learning (RL) is well known for being highly sensitive to hyperparameters, requiring practitioners substantial efforts to optimize them for the problem at hand. This also limits the applicability of RL in real-world scenarios. In recent years, the field of automated Reinforcement Learning (AutoRL) has grown in popularity by trying to address this issue. However, these approaches typically hinge on additional samples to select well-performing hyperparameters, hindering sample-efficiency and practicality. Furthermore, most AutoRL methods are heavily based on already existing A"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2405.16195","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.LG","submitted_at":"2024-05-25T11:57:43Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"0947eba339ee39583e1b6861b179f5cb7cdd7697eb86adde0a9b213cc61d1f52","abstract_canon_sha256":"b5584d52e073ce8c7fd48daba7615e13c4937222b0fcea016842f42ad98deed7"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:22:46.564058Z","signature_b64":"V9f1gA3bKVEa09RPrpqcB16Iwwh3eroKB0qZXLi/bwyqFIYIfF1zPCQYzLTBZb1RES4xITgxoafynL/W/gEYDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"209574a013bb595186cbce811ce4b631dab8692551e921ef0e0367c11db93283","last_reissued_at":"2026-07-05T10:22:46.563158Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:22:46.563158Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Adaptive $Q$-Network: On-the-fly Target Selection for Deep Reinforcement Learning","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Boris Belousov, Carlo D'Eramo, Fabian Wahren, Jan Peters, Th\\'eo Vincent","submitted_at":"2024-05-25T11:57:43Z","abstract_excerpt":"Deep Reinforcement Learning (RL) is well known for being highly sensitive to hyperparameters, requiring practitioners substantial efforts to optimize them for the problem at hand. This also limits the applicability of RL in real-world scenarios. In recent years, the field of automated Reinforcement Learning (AutoRL) has grown in popularity by trying to address this issue. However, these approaches typically hinge on additional samples to select well-performing hyperparameters, hindering sample-efficiency and practicality. Furthermore, most AutoRL methods are heavily based on already existing A"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2405.16195","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2405.16195/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2405.16195","created_at":"2026-07-05T10:22:46.563257+00:00"},{"alias_kind":"arxiv_version","alias_value":"2405.16195v3","created_at":"2026-07-05T10:22:46.563257+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2405.16195","created_at":"2026-07-05T10:22:46.563257+00:00"},{"alias_kind":"pith_short_12","alias_value":"ECKXJIATXNMV","created_at":"2026-07-05T10:22:46.563257+00:00"},{"alias_kind":"pith_short_16","alias_value":"ECKXJIATXNMVDBWL","created_at":"2026-07-05T10:22:46.563257+00:00"},{"alias_kind":"pith_short_8","alias_value":"ECKXJIAT","created_at":"2026-07-05T10:22:46.563257+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ECKXJIATXNMVDBWLZ2ARZZFWGH","json":"https://pith.science/pith/ECKXJIATXNMVDBWLZ2ARZZFWGH.json","graph_json":"https://pith.science/api/pith-number/ECKXJIATXNMVDBWLZ2ARZZFWGH/graph.json","events_json":"https://pith.science/api/pith-number/ECKXJIATXNMVDBWLZ2ARZZFWGH/events.json","paper":"https://pith.science/paper/ECKXJIAT"},"agent_actions":{"view_html":"https://pith.science/pith/ECKXJIATXNMVDBWLZ2ARZZFWGH","download_json":"https://pith.science/pith/ECKXJIATXNMVDBWLZ2ARZZFWGH.json","view_paper":"https://pith.science/paper/ECKXJIAT","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2405.16195&json=true","fetch_graph":"https://pith.science/api/pith-number/ECKXJIATXNMVDBWLZ2ARZZFWGH/graph.json","fetch_events":"https://pith.science/api/pith-number/ECKXJIATXNMVDBWLZ2ARZZFWGH/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ECKXJIATXNMVDBWLZ2ARZZFWGH/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ECKXJIATXNMVDBWLZ2ARZZFWGH/action/storage_attestation","attest_author":"https://pith.science/pith/ECKXJIATXNMVDBWLZ2ARZZFWGH/action/author_attestation","sign_citation":"https://pith.science/pith/ECKXJIATXNMVDBWLZ2ARZZFWGH/action/citation_signature","submit_replication":"https://pith.science/pith/ECKXJIATXNMVDBWLZ2ARZZFWGH/action/replication_record"}},"created_at":"2026-07-05T10:22:46.563257+00:00","updated_at":"2026-07-05T10:22:46.563257+00:00"}