{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:G7W6CVAJN3H2CEBASX7XMVYS3Z","short_pith_number":"pith:G7W6CVAJ","schema_version":"1.0","canonical_sha256":"37ede154096ecfa1102095ff765712de6cda3e4f5ee0af36bc14a51c5d8992e4","source":{"kind":"arxiv","id":"2306.01324","version":1},"attestation_state":"computed","paper":{"title":"Hyperparameters in Reinforcement Learning and How To Tune Them","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Marius Lindauer, Roberta Raileanu, Theresa Eimer","submitted_at":"2023-06-02T07:48:18Z","abstract_excerpt":"In order to improve reproducibility, deep reinforcement learning (RL) has been adopting better scientific practices such as standardized evaluation metrics and reporting. However, the process of hyperparameter optimization still varies widely across papers, which makes it challenging to compare RL algorithms fairly. In this paper, we show that hyperparameter choices in RL can significantly affect the agent's final performance and sample efficiency, and that the hyperparameter landscape can strongly depend on the tuning seed which may lead to overfitting. We therefore propose adopting establish"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2306.01324","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2023-06-02T07:48:18Z","cross_cats_sorted":[],"title_canon_sha256":"a926953176357adec1da341374dd59a5d6ac6610c71a11b6a5cccda8d8b41ea7","abstract_canon_sha256":"4c5f77a1dc5460c24f28ab9895390ce9d8e9284f6f3a98a9459674f0fec31766"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:16:52.555639Z","signature_b64":"/LviKFRdyfBWNZbBpndGWV8yd9+cS5Msg2Dpars+GhhZs2bQntviFWnEMcNIiT/h8/ZwBJrIuL9J3ijQ1ewVCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"37ede154096ecfa1102095ff765712de6cda3e4f5ee0af36bc14a51c5d8992e4","last_reissued_at":"2026-07-05T06:16:52.555291Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:16:52.555291Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Hyperparameters in Reinforcement Learning and How To Tune Them","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Marius Lindauer, Roberta Raileanu, Theresa Eimer","submitted_at":"2023-06-02T07:48:18Z","abstract_excerpt":"In order to improve reproducibility, deep reinforcement learning (RL) has been adopting better scientific practices such as standardized evaluation metrics and reporting. However, the process of hyperparameter optimization still varies widely across papers, which makes it challenging to compare RL algorithms fairly. In this paper, we show that hyperparameter choices in RL can significantly affect the agent's final performance and sample efficiency, and that the hyperparameter landscape can strongly depend on the tuning seed which may lead to overfitting. We therefore propose adopting establish"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2306.01324","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2306.01324/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2306.01324","created_at":"2026-07-05T06:16:52.555355+00:00"},{"alias_kind":"arxiv_version","alias_value":"2306.01324v1","created_at":"2026-07-05T06:16:52.555355+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2306.01324","created_at":"2026-07-05T06:16:52.555355+00:00"},{"alias_kind":"pith_short_12","alias_value":"G7W6CVAJN3H2","created_at":"2026-07-05T06:16:52.555355+00:00"},{"alias_kind":"pith_short_16","alias_value":"G7W6CVAJN3H2CEBA","created_at":"2026-07-05T06:16:52.555355+00:00"},{"alias_kind":"pith_short_8","alias_value":"G7W6CVAJ","created_at":"2026-07-05T06:16:52.555355+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.30719","citing_title":"When are LLMs Sufficient Policy Optimizers for Sequential RL Tasks?","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2605.30719","citing_title":"When are LLMs Sufficient Policy Optimizers for Sequential RL Tasks?","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2504.13541","citing_title":"Scalable Multi-Task Learning through Spiking Neural Networks with Adaptive Task-Switching Policy for Intelligent Autonomous Agents","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01567","citing_title":"Feedback-Normalized Developer Memory for Reinforcement-Learning Coding Agents: A Safety-Gated MCP Architecture","ref_index":5,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/G7W6CVAJN3H2CEBASX7XMVYS3Z","json":"https://pith.science/pith/G7W6CVAJN3H2CEBASX7XMVYS3Z.json","graph_json":"https://pith.science/api/pith-number/G7W6CVAJN3H2CEBASX7XMVYS3Z/graph.json","events_json":"https://pith.science/api/pith-number/G7W6CVAJN3H2CEBASX7XMVYS3Z/events.json","paper":"https://pith.science/paper/G7W6CVAJ"},"agent_actions":{"view_html":"https://pith.science/pith/G7W6CVAJN3H2CEBASX7XMVYS3Z","download_json":"https://pith.science/pith/G7W6CVAJN3H2CEBASX7XMVYS3Z.json","view_paper":"https://pith.science/paper/G7W6CVAJ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2306.01324&json=true","fetch_graph":"https://pith.science/api/pith-number/G7W6CVAJN3H2CEBASX7XMVYS3Z/graph.json","fetch_events":"https://pith.science/api/pith-number/G7W6CVAJN3H2CEBASX7XMVYS3Z/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/G7W6CVAJN3H2CEBASX7XMVYS3Z/action/timestamp_anchor","attest_storage":"https://pith.science/pith/G7W6CVAJN3H2CEBASX7XMVYS3Z/action/storage_attestation","attest_author":"https://pith.science/pith/G7W6CVAJN3H2CEBASX7XMVYS3Z/action/author_attestation","sign_citation":"https://pith.science/pith/G7W6CVAJN3H2CEBASX7XMVYS3Z/action/citation_signature","submit_replication":"https://pith.science/pith/G7W6CVAJN3H2CEBASX7XMVYS3Z/action/replication_record"}},"created_at":"2026-07-05T06:16:52.555355+00:00","updated_at":"2026-07-05T06:16:52.555355+00:00"}