{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:O7HNYNUNDDDLTH6MKWB7CYARK4","short_pith_number":"pith:O7HNYNUN","schema_version":"1.0","canonical_sha256":"77cedc368d18c6b99fcc5583f16011570fc55c668a67249d13fbaba3bc09d894","source":{"kind":"arxiv","id":"2407.07082","version":3},"attestation_state":"computed","paper":{"title":"Can Learned Optimization Make Reinforcement Learning Less Difficult?","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Alexander David Goldie, Chris Lu, Jakob Nicolaus Foerster, Matthew Thomas Jackson, Shimon Whiteson","submitted_at":"2024-07-09T17:55:23Z","abstract_excerpt":"While reinforcement learning (RL) holds great potential for decision making in the real world, it suffers from a number of unique difficulties which often need specific consideration. In particular: it is highly non-stationary; suffers from high degrees of plasticity loss; and requires exploration to prevent premature convergence to local optima and maximize return. In this paper, we consider whether learned optimization can help overcome these problems. Our method, Learned Optimization for Plasticity, Exploration and Non-stationarity (OPEN), meta-learns an update rule whose input features and"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2407.07082","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-07-09T17:55:23Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"1d43a1d1c2486750a971c3300dfb1145f19599e00ea701627b4e9ad585df1a1e","abstract_canon_sha256":"011f497a359b4e59a06dd599321c9759503f30b283f1cfda7f118fb0c87403ff"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:49:07.227262Z","signature_b64":"vc4tAHPs+bo1uLuJWst6W6yeQ9Njq/IoNYF/6qUM/xVwniLVzU8GC2PWyYIyC3l6lLcxFyvKxjOyK/GAkWuJDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"77cedc368d18c6b99fcc5583f16011570fc55c668a67249d13fbaba3bc09d894","last_reissued_at":"2026-07-05T10:49:07.226724Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:49:07.226724Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Can Learned Optimization Make Reinforcement Learning Less Difficult?","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Alexander David Goldie, Chris Lu, Jakob Nicolaus Foerster, Matthew Thomas Jackson, Shimon Whiteson","submitted_at":"2024-07-09T17:55:23Z","abstract_excerpt":"While reinforcement learning (RL) holds great potential for decision making in the real world, it suffers from a number of unique difficulties which often need specific consideration. In particular: it is highly non-stationary; suffers from high degrees of plasticity loss; and requires exploration to prevent premature convergence to local optima and maximize return. In this paper, we consider whether learned optimization can help overcome these problems. Our method, Learned Optimization for Plasticity, Exploration and Non-stationarity (OPEN), meta-learns an update rule whose input features and"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2407.07082","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2407.07082/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2407.07082","created_at":"2026-07-05T10:49:07.226789+00:00"},{"alias_kind":"arxiv_version","alias_value":"2407.07082v3","created_at":"2026-07-05T10:49:07.226789+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2407.07082","created_at":"2026-07-05T10:49:07.226789+00:00"},{"alias_kind":"pith_short_12","alias_value":"O7HNYNUNDDDL","created_at":"2026-07-05T10:49:07.226789+00:00"},{"alias_kind":"pith_short_16","alias_value":"O7HNYNUNDDDLTH6M","created_at":"2026-07-05T10:49:07.226789+00:00"},{"alias_kind":"pith_short_8","alias_value":"O7HNYNUN","created_at":"2026-07-05T10:49:07.226789+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.29559","citing_title":"LEMUR: Learning to Align with Multi-Objective Reinforcement Learning from Preference Feedback","ref_index":265,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/O7HNYNUNDDDLTH6MKWB7CYARK4","json":"https://pith.science/pith/O7HNYNUNDDDLTH6MKWB7CYARK4.json","graph_json":"https://pith.science/api/pith-number/O7HNYNUNDDDLTH6MKWB7CYARK4/graph.json","events_json":"https://pith.science/api/pith-number/O7HNYNUNDDDLTH6MKWB7CYARK4/events.json","paper":"https://pith.science/paper/O7HNYNUN"},"agent_actions":{"view_html":"https://pith.science/pith/O7HNYNUNDDDLTH6MKWB7CYARK4","download_json":"https://pith.science/pith/O7HNYNUNDDDLTH6MKWB7CYARK4.json","view_paper":"https://pith.science/paper/O7HNYNUN","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2407.07082&json=true","fetch_graph":"https://pith.science/api/pith-number/O7HNYNUNDDDLTH6MKWB7CYARK4/graph.json","fetch_events":"https://pith.science/api/pith-number/O7HNYNUNDDDLTH6MKWB7CYARK4/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/O7HNYNUNDDDLTH6MKWB7CYARK4/action/timestamp_anchor","attest_storage":"https://pith.science/pith/O7HNYNUNDDDLTH6MKWB7CYARK4/action/storage_attestation","attest_author":"https://pith.science/pith/O7HNYNUNDDDLTH6MKWB7CYARK4/action/author_attestation","sign_citation":"https://pith.science/pith/O7HNYNUNDDDLTH6MKWB7CYARK4/action/citation_signature","submit_replication":"https://pith.science/pith/O7HNYNUNDDDLTH6MKWB7CYARK4/action/replication_record"}},"created_at":"2026-07-05T10:49:07.226789+00:00","updated_at":"2026-07-05T10:49:07.226789+00:00"}