{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:4HCD5KXUORFYXJITC7RNU7LDZM","short_pith_number":"pith:4HCD5KXU","schema_version":"1.0","canonical_sha256":"e1c43eaaf4744b8ba51317e2da7d63cb07685517adfbb9cf6a90ffd7f5d16f88","source":{"kind":"arxiv","id":"2403.05996","version":3},"attestation_state":"computed","paper":{"title":"Dissecting Deep RL with High Update Ratios: Combatting Value Divergence","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Amir-massoud Farahmand, Claas Voelcker, Eric Eaton, Igor Gilitschenski, Marcel Hussing","submitted_at":"2024-03-09T19:56:40Z","abstract_excerpt":"We show that deep reinforcement learning algorithms can retain their ability to learn without resetting network parameters in settings where the number of gradient updates greatly exceeds the number of environment samples by combatting value function divergence. Under large update-to-data ratios, a recent study by Nikishin et al. (2022) suggested the emergence of a primacy bias, in which agents overfit early interactions and downplay later experience, impairing their ability to learn. In this work, we investigate the phenomena leading to the primacy bias. We inspect the early stages of trainin"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2403.05996","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-03-09T19:56:40Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"c5150ce4a2dcf2d5c8ab67fb771b3dd9151fd96b766318193996033140e49794","abstract_canon_sha256":"41021f4ec0c73c07fa670a2b99b5bf05a3e7199c408891e0b1125fd3a6000058"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:51:49.139868Z","signature_b64":"WBs9HFnDsGNX+07Nvz3EFvRIHbgBJbQNKvyUACLn8sH+gAmjXMY6crSWPYgGV2I4FK0qtmxpqZObgDz4LT7+Bw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"e1c43eaaf4744b8ba51317e2da7d63cb07685517adfbb9cf6a90ffd7f5d16f88","last_reissued_at":"2026-07-05T08:51:49.139471Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:51:49.139471Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Dissecting Deep RL with High Update Ratios: Combatting Value Divergence","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Amir-massoud Farahmand, Claas Voelcker, Eric Eaton, Igor Gilitschenski, Marcel Hussing","submitted_at":"2024-03-09T19:56:40Z","abstract_excerpt":"We show that deep reinforcement learning algorithms can retain their ability to learn without resetting network parameters in settings where the number of gradient updates greatly exceeds the number of environment samples by combatting value function divergence. Under large update-to-data ratios, a recent study by Nikishin et al. (2022) suggested the emergence of a primacy bias, in which agents overfit early interactions and downplay later experience, impairing their ability to learn. In this work, we investigate the phenomena leading to the primacy bias. We inspect the early stages of trainin"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2403.05996","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2403.05996/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2403.05996","created_at":"2026-07-05T08:51:49.139524+00:00"},{"alias_kind":"arxiv_version","alias_value":"2403.05996v3","created_at":"2026-07-05T08:51:49.139524+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2403.05996","created_at":"2026-07-05T08:51:49.139524+00:00"},{"alias_kind":"pith_short_12","alias_value":"4HCD5KXUORFY","created_at":"2026-07-05T08:51:49.139524+00:00"},{"alias_kind":"pith_short_16","alias_value":"4HCD5KXUORFYXJIT","created_at":"2026-07-05T08:51:49.139524+00:00"},{"alias_kind":"pith_short_8","alias_value":"4HCD5KXU","created_at":"2026-07-05T08:51:49.139524+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2603.04333","citing_title":"What Does Flow Matching Bring To TD Learning?","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2604.23073","citing_title":"RL Token: Bootstrapping Online RL with Vision-Language-Action Models","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2604.20381","citing_title":"Distributional Value Estimation Without Target Networks for Robust Quality-Diversity","ref_index":24,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/4HCD5KXUORFYXJITC7RNU7LDZM","json":"https://pith.science/pith/4HCD5KXUORFYXJITC7RNU7LDZM.json","graph_json":"https://pith.science/api/pith-number/4HCD5KXUORFYXJITC7RNU7LDZM/graph.json","events_json":"https://pith.science/api/pith-number/4HCD5KXUORFYXJITC7RNU7LDZM/events.json","paper":"https://pith.science/paper/4HCD5KXU"},"agent_actions":{"view_html":"https://pith.science/pith/4HCD5KXUORFYXJITC7RNU7LDZM","download_json":"https://pith.science/pith/4HCD5KXUORFYXJITC7RNU7LDZM.json","view_paper":"https://pith.science/paper/4HCD5KXU","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2403.05996&json=true","fetch_graph":"https://pith.science/api/pith-number/4HCD5KXUORFYXJITC7RNU7LDZM/graph.json","fetch_events":"https://pith.science/api/pith-number/4HCD5KXUORFYXJITC7RNU7LDZM/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/4HCD5KXUORFYXJITC7RNU7LDZM/action/timestamp_anchor","attest_storage":"https://pith.science/pith/4HCD5KXUORFYXJITC7RNU7LDZM/action/storage_attestation","attest_author":"https://pith.science/pith/4HCD5KXUORFYXJITC7RNU7LDZM/action/author_attestation","sign_citation":"https://pith.science/pith/4HCD5KXUORFYXJITC7RNU7LDZM/action/citation_signature","submit_replication":"https://pith.science/pith/4HCD5KXUORFYXJITC7RNU7LDZM/action/replication_record"}},"created_at":"2026-07-05T08:51:49.139524+00:00","updated_at":"2026-07-05T08:51:49.139524+00:00"}