{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2023:G7W6CVAJN3H2CEBASX7XMVYS3Z","short_pith_number":"pith:G7W6CVAJ","canonical_record":{"source":{"id":"2306.01324","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2023-06-02T07:48:18Z","cross_cats_sorted":[],"title_canon_sha256":"a926953176357adec1da341374dd59a5d6ac6610c71a11b6a5cccda8d8b41ea7","abstract_canon_sha256":"4c5f77a1dc5460c24f28ab9895390ce9d8e9284f6f3a98a9459674f0fec31766"},"schema_version":"1.0"},"canonical_sha256":"37ede154096ecfa1102095ff765712de6cda3e4f5ee0af36bc14a51c5d8992e4","source":{"kind":"arxiv","id":"2306.01324","version":1},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2306.01324","created_at":"2026-07-05T06:16:52Z"},{"alias_kind":"arxiv_version","alias_value":"2306.01324v1","created_at":"2026-07-05T06:16:52Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2306.01324","created_at":"2026-07-05T06:16:52Z"},{"alias_kind":"pith_short_12","alias_value":"G7W6CVAJN3H2","created_at":"2026-07-05T06:16:52Z"},{"alias_kind":"pith_short_16","alias_value":"G7W6CVAJN3H2CEBA","created_at":"2026-07-05T06:16:52Z"},{"alias_kind":"pith_short_8","alias_value":"G7W6CVAJ","created_at":"2026-07-05T06:16:52Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2023:G7W6CVAJN3H2CEBASX7XMVYS3Z","target":"record","payload":{"canonical_record":{"source":{"id":"2306.01324","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2023-06-02T07:48:18Z","cross_cats_sorted":[],"title_canon_sha256":"a926953176357adec1da341374dd59a5d6ac6610c71a11b6a5cccda8d8b41ea7","abstract_canon_sha256":"4c5f77a1dc5460c24f28ab9895390ce9d8e9284f6f3a98a9459674f0fec31766"},"schema_version":"1.0"},"canonical_sha256":"37ede154096ecfa1102095ff765712de6cda3e4f5ee0af36bc14a51c5d8992e4","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:16:52.555639Z","signature_b64":"/LviKFRdyfBWNZbBpndGWV8yd9+cS5Msg2Dpars+GhhZs2bQntviFWnEMcNIiT/h8/ZwBJrIuL9J3ijQ1ewVCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"37ede154096ecfa1102095ff765712de6cda3e4f5ee0af36bc14a51c5d8992e4","last_reissued_at":"2026-07-05T06:16:52.555291Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:16:52.555291Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2306.01324","source_version":1,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T06:16:52Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"0c3j0ubRTFqGXhtLHZbb/dGDxJ3Vua5VbmRi/npiyG/1EhicRSqg58CX9DCMpOfSrcN12GD+cq4Uyaf3ExHOBQ==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-11T10:18:21.637303Z"},"content_sha256":"ac3c16f5674d27fa55205d23b22cfc46aa4f765bb52b0414a99ba72a4da0eb69","schema_version":"1.0","event_id":"sha256:ac3c16f5674d27fa55205d23b22cfc46aa4f765bb52b0414a99ba72a4da0eb69"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2023:G7W6CVAJN3H2CEBASX7XMVYS3Z","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Hyperparameters in Reinforcement Learning and How To Tune Them","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Marius Lindauer, Roberta Raileanu, Theresa Eimer","submitted_at":"2023-06-02T07:48:18Z","abstract_excerpt":"In order to improve reproducibility, deep reinforcement learning (RL) has been adopting better scientific practices such as standardized evaluation metrics and reporting. However, the process of hyperparameter optimization still varies widely across papers, which makes it challenging to compare RL algorithms fairly. In this paper, we show that hyperparameter choices in RL can significantly affect the agent's final performance and sample efficiency, and that the hyperparameter landscape can strongly depend on the tuning seed which may lead to overfitting. We therefore propose adopting establish"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2306.01324","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2306.01324/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T06:16:52Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"+pe8l0ElxXV+jSu2uVa3yEaZCfa0YJPsNdztfJT07+zOzECR3XqvIlU0a/1E68cInqopv47KUrgdjgkp7E0EAQ==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-11T10:18:21.637636Z"},"content_sha256":"d57fbb993ccb9ed45ed2f810b88ad509dffc32a79e544deabdcc053adea34b47","schema_version":"1.0","event_id":"sha256:d57fbb993ccb9ed45ed2f810b88ad509dffc32a79e544deabdcc053adea34b47"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/G7W6CVAJN3H2CEBASX7XMVYS3Z/bundle.json","state_url":"https://pith.science/pith/G7W6CVAJN3H2CEBASX7XMVYS3Z/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/G7W6CVAJN3H2CEBASX7XMVYS3Z/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-11T10:18:21Z","links":{"resolver":"https://pith.science/pith/G7W6CVAJN3H2CEBASX7XMVYS3Z","bundle":"https://pith.science/pith/G7W6CVAJN3H2CEBASX7XMVYS3Z/bundle.json","state":"https://pith.science/pith/G7W6CVAJN3H2CEBASX7XMVYS3Z/state.json","well_known_bundle":"https://pith.science/.well-known/pith/G7W6CVAJN3H2CEBASX7XMVYS3Z/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2023:G7W6CVAJN3H2CEBASX7XMVYS3Z","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"4c5f77a1dc5460c24f28ab9895390ce9d8e9284f6f3a98a9459674f0fec31766","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2023-06-02T07:48:18Z","title_canon_sha256":"a926953176357adec1da341374dd59a5d6ac6610c71a11b6a5cccda8d8b41ea7"},"schema_version":"1.0","source":{"id":"2306.01324","kind":"arxiv","version":1}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2306.01324","created_at":"2026-07-05T06:16:52Z"},{"alias_kind":"arxiv_version","alias_value":"2306.01324v1","created_at":"2026-07-05T06:16:52Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2306.01324","created_at":"2026-07-05T06:16:52Z"},{"alias_kind":"pith_short_12","alias_value":"G7W6CVAJN3H2","created_at":"2026-07-05T06:16:52Z"},{"alias_kind":"pith_short_16","alias_value":"G7W6CVAJN3H2CEBA","created_at":"2026-07-05T06:16:52Z"},{"alias_kind":"pith_short_8","alias_value":"G7W6CVAJ","created_at":"2026-07-05T06:16:52Z"}],"graph_snapshots":[{"event_id":"sha256:d57fbb993ccb9ed45ed2f810b88ad509dffc32a79e544deabdcc053adea34b47","target":"graph","created_at":"2026-07-05T06:16:52Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2306.01324/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"In order to improve reproducibility, deep reinforcement learning (RL) has been adopting better scientific practices such as standardized evaluation metrics and reporting. However, the process of hyperparameter optimization still varies widely across papers, which makes it challenging to compare RL algorithms fairly. In this paper, we show that hyperparameter choices in RL can significantly affect the agent's final performance and sample efficiency, and that the hyperparameter landscape can strongly depend on the tuning seed which may lead to overfitting. We therefore propose adopting establish","authors_text":"Marius Lindauer, Roberta Raileanu, Theresa Eimer","cross_cats":[],"headline":"","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2023-06-02T07:48:18Z","title":"Hyperparameters in Reinforcement Learning and How To Tune Them"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2306.01324","kind":"arxiv","version":1},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:ac3c16f5674d27fa55205d23b22cfc46aa4f765bb52b0414a99ba72a4da0eb69","target":"record","created_at":"2026-07-05T06:16:52Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"4c5f77a1dc5460c24f28ab9895390ce9d8e9284f6f3a98a9459674f0fec31766","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2023-06-02T07:48:18Z","title_canon_sha256":"a926953176357adec1da341374dd59a5d6ac6610c71a11b6a5cccda8d8b41ea7"},"schema_version":"1.0","source":{"id":"2306.01324","kind":"arxiv","version":1}},"canonical_sha256":"37ede154096ecfa1102095ff765712de6cda3e4f5ee0af36bc14a51c5d8992e4","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"37ede154096ecfa1102095ff765712de6cda3e4f5ee0af36bc14a51c5d8992e4","first_computed_at":"2026-07-05T06:16:52.555291Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T06:16:52.555291Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"/LviKFRdyfBWNZbBpndGWV8yd9+cS5Msg2Dpars+GhhZs2bQntviFWnEMcNIiT/h8/ZwBJrIuL9J3ijQ1ewVCw==","signature_status":"signed_v1","signed_at":"2026-07-05T06:16:52.555639Z","signed_message":"canonical_sha256_bytes"},"source_id":"2306.01324","source_kind":"arxiv","source_version":1}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:ac3c16f5674d27fa55205d23b22cfc46aa4f765bb52b0414a99ba72a4da0eb69","sha256:d57fbb993ccb9ed45ed2f810b88ad509dffc32a79e544deabdcc053adea34b47"],"state_sha256":"5eff8516251adb8adec0a91c9ef0dbfd253e6f7ec3f7a949b367ff2fa64eac7f"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"c15XUhhq4PYwyppuffKkRCdOiBWy1dPMpXJB3RnMWPAeXdDJWnv8itN/4nmWevTnkL81ElnvXYUnDSHAgVO/AA==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-11T10:18:21.640298Z","bundle_sha256":"6c20e67d3f1754822fc8e4a07acbee2c70f93d24dc240a83e6d79ba54ac87fe1"}}