{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:EYTSET24SA7E7F4EXAX5VFHZSG","short_pith_number":"pith:EYTSET24","schema_version":"1.0","canonical_sha256":"2627224f5c903e4f9784b82fda94f991b45ccf8f48c2ed8c39244db623e474f6","source":{"kind":"arxiv","id":"2407.04811","version":6},"attestation_state":"computed","paper":{"title":"Simplifying Deep Temporal Difference Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Bartomeu Pou, Benjamin Ellis, Ivan Masmitja, Jakob Nicolaus Foerster, Mario Martin, Matteo Gallici, Mattie Fellows","submitted_at":"2024-07-05T18:49:07Z","abstract_excerpt":"Q-learning played a foundational role in the field reinforcement learning (RL). However, TD algorithms with off-policy data, such as Q-learning, or nonlinear function approximation like deep neural networks require several additional tricks to stabilise training, primarily a large replay buffer and target networks. Unfortunately, the delayed updating of frozen network parameters in the target network harms the sample efficiency and, similarly, the large replay buffer introduces memory and implementation overheads. In this paper, we investigate whether it is possible to accelerate and simplify "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2407.04811","kind":"arxiv","version":6},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-07-05T18:49:07Z","cross_cats_sorted":[],"title_canon_sha256":"434926274aeb1a57346d2b448a9dc5a7d9d364bd7d08a5342bb6e7bf40d9a57d","abstract_canon_sha256":"a092048e00099d3176b38dab36189ab3ff221c909b9eb5fed1c7503627b2b85d"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:52:12.670712Z","signature_b64":"mKuUIzyh5NZYVg8xGruPSfMHVLQE2HVD6IT7Y09XbCIULUkMLmXYP5OddPXVlGzQ+UwyNsKAYUOr2fjK2prODQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"2627224f5c903e4f9784b82fda94f991b45ccf8f48c2ed8c39244db623e474f6","last_reissued_at":"2026-07-05T10:52:12.670112Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:52:12.670112Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Simplifying Deep Temporal Difference Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Bartomeu Pou, Benjamin Ellis, Ivan Masmitja, Jakob Nicolaus Foerster, Mario Martin, Matteo Gallici, Mattie Fellows","submitted_at":"2024-07-05T18:49:07Z","abstract_excerpt":"Q-learning played a foundational role in the field reinforcement learning (RL). However, TD algorithms with off-policy data, such as Q-learning, or nonlinear function approximation like deep neural networks require several additional tricks to stabilise training, primarily a large replay buffer and target networks. Unfortunately, the delayed updating of frozen network parameters in the target network harms the sample efficiency and, similarly, the large replay buffer introduces memory and implementation overheads. In this paper, we investigate whether it is possible to accelerate and simplify "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2407.04811","kind":"arxiv","version":6},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2407.04811/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2407.04811","created_at":"2026-07-05T10:52:12.670173+00:00"},{"alias_kind":"arxiv_version","alias_value":"2407.04811v6","created_at":"2026-07-05T10:52:12.670173+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2407.04811","created_at":"2026-07-05T10:52:12.670173+00:00"},{"alias_kind":"pith_short_12","alias_value":"EYTSET24SA7E","created_at":"2026-07-05T10:52:12.670173+00:00"},{"alias_kind":"pith_short_16","alias_value":"EYTSET24SA7E7F4E","created_at":"2026-07-05T10:52:12.670173+00:00"},{"alias_kind":"pith_short_8","alias_value":"EYTSET24","created_at":"2026-07-05T10:52:12.670173+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":12,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.21943","citing_title":"Modularized Reinforcement Learning on LLMs: From MDP Creation to Exploration and Learning","ref_index":53,"is_internal_anchor":false},{"citing_arxiv_id":"2605.24975","citing_title":"Bridging the Gap: Enabling Soft Actor Critic for High Performance Legged Locomotion","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2605.21557","citing_title":"Scalable Reinforcement Learning via Adaptive Batch Scaling","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2606.00880","citing_title":"Task diversity produces systematic transfer but inhibits continual reinforcement learning","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2606.04735","citing_title":"Trace-Mediated Peak Bias: Bridging Temporal Credit Assignment and Cognitive Heuristics in Deep Reinforcement Learning","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2602.01665","citing_title":"TABX: A High-Throughput Sandbox Battle Simulator for Multi-Agent Reinforcement Learning","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2605.23551","citing_title":"Goal-Conditioned Agents that Learn Everything All at Once","ref_index":68,"is_internal_anchor":false},{"citing_arxiv_id":"2411.04832","citing_title":"Plasticity Loss in Deep Reinforcement Learning: A Survey","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2605.21557","citing_title":"Scalable Reinforcement Learning via Adaptive Batch Scaling","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2604.04539","citing_title":"FlashSAC: Fast and Stable Off-Policy Reinforcement Learning for High-Dimensional Robot Control","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2604.27162","citing_title":"A High-Throughput Compute-Efficient POMDP Hide-And-Seek-Engine (HASE) for Multi-Agent Operations","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2604.04539","citing_title":"FlashSAC: Fast and Stable Off-Policy Reinforcement Learning for High-Dimensional Robot Control","ref_index":18,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/EYTSET24SA7E7F4EXAX5VFHZSG","json":"https://pith.science/pith/EYTSET24SA7E7F4EXAX5VFHZSG.json","graph_json":"https://pith.science/api/pith-number/EYTSET24SA7E7F4EXAX5VFHZSG/graph.json","events_json":"https://pith.science/api/pith-number/EYTSET24SA7E7F4EXAX5VFHZSG/events.json","paper":"https://pith.science/paper/EYTSET24"},"agent_actions":{"view_html":"https://pith.science/pith/EYTSET24SA7E7F4EXAX5VFHZSG","download_json":"https://pith.science/pith/EYTSET24SA7E7F4EXAX5VFHZSG.json","view_paper":"https://pith.science/paper/EYTSET24","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2407.04811&json=true","fetch_graph":"https://pith.science/api/pith-number/EYTSET24SA7E7F4EXAX5VFHZSG/graph.json","fetch_events":"https://pith.science/api/pith-number/EYTSET24SA7E7F4EXAX5VFHZSG/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/EYTSET24SA7E7F4EXAX5VFHZSG/action/timestamp_anchor","attest_storage":"https://pith.science/pith/EYTSET24SA7E7F4EXAX5VFHZSG/action/storage_attestation","attest_author":"https://pith.science/pith/EYTSET24SA7E7F4EXAX5VFHZSG/action/author_attestation","sign_citation":"https://pith.science/pith/EYTSET24SA7E7F4EXAX5VFHZSG/action/citation_signature","submit_replication":"https://pith.science/pith/EYTSET24SA7E7F4EXAX5VFHZSG/action/replication_record"}},"created_at":"2026-07-05T10:52:12.670173+00:00","updated_at":"2026-07-05T10:52:12.670173+00:00"}