{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2018:ZVBMYROB2TA3SIXAMLJDGJ6BXI","short_pith_number":"pith:ZVBMYROB","schema_version":"1.0","canonical_sha256":"cd42cc45c1d4c1b922e062d23327c1ba2ff286ecc747b629e784c7240cd98b1d","source":{"kind":"arxiv","id":"1810.00123","version":3},"attestation_state":"computed","paper":{"title":"Generalization and Regularization in DQN","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","stat.ML"],"primary_cat":"cs.LG","authors_text":"Jesse Farebrother, Marlos C. Machado, Michael Bowling","submitted_at":"2018-09-29T00:52:34Z","abstract_excerpt":"Deep reinforcement learning algorithms have shown an impressive ability to learn complex control policies in high-dimensional tasks. However, despite the ever-increasing performance on popular benchmarks, policies learned by deep reinforcement learning algorithms can struggle to generalize when evaluated in remarkably similar environments. In this paper we propose a protocol to evaluate generalization in reinforcement learning through different modes of Atari 2600 games. With that protocol we assess the generalization capabilities of DQN, one of the most traditional deep reinforcement learning"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"1810.00123","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2018-09-29T00:52:34Z","cross_cats_sorted":["cs.AI","stat.ML"],"title_canon_sha256":"e722447df5599d1c0340059f478306113b029bd51559a8bf28472e5d11b9fef6","abstract_canon_sha256":"447ea4407ec498944a9267428fe0ede408923e43ebf09840ec079c6f6ef3a175"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T00:34:13.390600Z","signature_b64":"LCC5Xkjy6SIj9hqmfYeLESPO7YxkJIyoE1Oqzkmo6KCzbRuCbI7QGrgrYdyeYxmJUTTIpEouhNq9fVygRhOrAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"cd42cc45c1d4c1b922e062d23327c1ba2ff286ecc747b629e784c7240cd98b1d","last_reissued_at":"2026-07-05T00:34:13.390096Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T00:34:13.390096Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Generalization and Regularization in DQN","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","stat.ML"],"primary_cat":"cs.LG","authors_text":"Jesse Farebrother, Marlos C. Machado, Michael Bowling","submitted_at":"2018-09-29T00:52:34Z","abstract_excerpt":"Deep reinforcement learning algorithms have shown an impressive ability to learn complex control policies in high-dimensional tasks. However, despite the ever-increasing performance on popular benchmarks, policies learned by deep reinforcement learning algorithms can struggle to generalize when evaluated in remarkably similar environments. In this paper we propose a protocol to evaluate generalization in reinforcement learning through different modes of Atari 2600 games. With that protocol we assess the generalization capabilities of DQN, one of the most traditional deep reinforcement learning"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1810.00123","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/1810.00123/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"1810.00123","created_at":"2026-07-05T00:34:13.390159+00:00"},{"alias_kind":"arxiv_version","alias_value":"1810.00123v3","created_at":"2026-07-05T00:34:13.390159+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1810.00123","created_at":"2026-07-05T00:34:13.390159+00:00"},{"alias_kind":"pith_short_12","alias_value":"ZVBMYROB2TA3","created_at":"2026-07-05T00:34:13.390159+00:00"},{"alias_kind":"pith_short_16","alias_value":"ZVBMYROB2TA3SIXA","created_at":"2026-07-05T00:34:13.390159+00:00"},{"alias_kind":"pith_short_8","alias_value":"ZVBMYROB","created_at":"2026-07-05T00:34:13.390159+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":8,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.27348","citing_title":"Bridging Performance and Generalization in Reinforcement Learning for Agile Flight","ref_index":42,"is_internal_anchor":false},{"citing_arxiv_id":"1906.09781","citing_title":"In Hindsight: A Smooth Reward for Steady Exploration","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"1907.01475","citing_title":"Generalizing from a few environments in safety-critical reinforcement learning","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"1907.02050","citing_title":"Reasoning and Generalization in RL: A Tool Use Perspective","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2605.21214","citing_title":"Behavior-Consistent Deep Reinforcement Learning","ref_index":225,"is_internal_anchor":false},{"citing_arxiv_id":"2605.21214","citing_title":"Behavior-Consistent Deep Reinforcement Learning","ref_index":225,"is_internal_anchor":false},{"citing_arxiv_id":"2210.10760","citing_title":"Scaling Laws for Reward Model Overoptimization","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2309.07864","citing_title":"The Rise and Potential of Large Language Model Based Agents: A Survey","ref_index":73,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ZVBMYROB2TA3SIXAMLJDGJ6BXI","json":"https://pith.science/pith/ZVBMYROB2TA3SIXAMLJDGJ6BXI.json","graph_json":"https://pith.science/api/pith-number/ZVBMYROB2TA3SIXAMLJDGJ6BXI/graph.json","events_json":"https://pith.science/api/pith-number/ZVBMYROB2TA3SIXAMLJDGJ6BXI/events.json","paper":"https://pith.science/paper/ZVBMYROB"},"agent_actions":{"view_html":"https://pith.science/pith/ZVBMYROB2TA3SIXAMLJDGJ6BXI","download_json":"https://pith.science/pith/ZVBMYROB2TA3SIXAMLJDGJ6BXI.json","view_paper":"https://pith.science/paper/ZVBMYROB","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=1810.00123&json=true","fetch_graph":"https://pith.science/api/pith-number/ZVBMYROB2TA3SIXAMLJDGJ6BXI/graph.json","fetch_events":"https://pith.science/api/pith-number/ZVBMYROB2TA3SIXAMLJDGJ6BXI/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ZVBMYROB2TA3SIXAMLJDGJ6BXI/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ZVBMYROB2TA3SIXAMLJDGJ6BXI/action/storage_attestation","attest_author":"https://pith.science/pith/ZVBMYROB2TA3SIXAMLJDGJ6BXI/action/author_attestation","sign_citation":"https://pith.science/pith/ZVBMYROB2TA3SIXAMLJDGJ6BXI/action/citation_signature","submit_replication":"https://pith.science/pith/ZVBMYROB2TA3SIXAMLJDGJ6BXI/action/replication_record"}},"created_at":"2026-07-05T00:34:13.390159+00:00","updated_at":"2026-07-05T00:34:13.390159+00:00"}