{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2021:D5FRKACD75NVLIHYMQLWCMKIU7","short_pith_number":"pith:D5FRKACD","canonical_record":{"source":{"id":"2112.04716","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2021-12-09T06:01:01Z","cross_cats_sorted":[],"title_canon_sha256":"49e244f0893b73e12e89e08c3aa394ed04254404461809a0400008bdc3967947","abstract_canon_sha256":"58a8e80193b842693b53e006f637a5ac674c14d9146294156e4c5ab6e20a3507"},"schema_version":"1.0"},"canonical_sha256":"1f4b150043ff5b55a0f86417613148a7e83d1093456709262c7d8425384dcfff","source":{"kind":"arxiv","id":"2112.04716","version":1},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2112.04716","created_at":"2026-07-05T03:39:22Z"},{"alias_kind":"arxiv_version","alias_value":"2112.04716v1","created_at":"2026-07-05T03:39:22Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2112.04716","created_at":"2026-07-05T03:39:22Z"},{"alias_kind":"pith_short_12","alias_value":"D5FRKACD75NV","created_at":"2026-07-05T03:39:22Z"},{"alias_kind":"pith_short_16","alias_value":"D5FRKACD75NVLIHY","created_at":"2026-07-05T03:39:22Z"},{"alias_kind":"pith_short_8","alias_value":"D5FRKACD","created_at":"2026-07-05T03:39:22Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2021:D5FRKACD75NVLIHYMQLWCMKIU7","target":"record","payload":{"canonical_record":{"source":{"id":"2112.04716","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2021-12-09T06:01:01Z","cross_cats_sorted":[],"title_canon_sha256":"49e244f0893b73e12e89e08c3aa394ed04254404461809a0400008bdc3967947","abstract_canon_sha256":"58a8e80193b842693b53e006f637a5ac674c14d9146294156e4c5ab6e20a3507"},"schema_version":"1.0"},"canonical_sha256":"1f4b150043ff5b55a0f86417613148a7e83d1093456709262c7d8425384dcfff","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T03:39:22.895365Z","signature_b64":"J5X48pnKj1F4x4xwyyq4EgyIzQ3U575bayaUdAyb+RUEeV9TtC8Xl3SkgAU8OqQ3j+Gt5OPIkrREDFWyR6tGBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"1f4b150043ff5b55a0f86417613148a7e83d1093456709262c7d8425384dcfff","last_reissued_at":"2026-07-05T03:39:22.894985Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T03:39:22.894985Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2112.04716","source_version":1,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T03:39:22Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"5oAvZRVeWUX11i8qjVLIow9saByOqyHhu7xVO0NLCV9XeCHESYob4EgOMHZL7JXg80BwovrqqJ4K+2J900wfDA==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-05T10:34:06.181799Z"},"content_sha256":"1e3697b6c4659e2c7307b8aae0322210e0038fe768782f7e5c8d8ca429beebfb","schema_version":"1.0","event_id":"sha256:1e3697b6c4659e2c7307b8aae0322210e0038fe768782f7e5c8d8ca429beebfb"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2021:D5FRKACD75NVLIHYMQLWCMKIU7","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"DR3: Value-Based Deep Reinforcement Learning Requires Explicit Regularization","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Aaron Courville, Aviral Kumar, George Tucker, Rishabh Agarwal, Sergey Levine, Tengyu Ma","submitted_at":"2021-12-09T06:01:01Z","abstract_excerpt":"Despite overparameterization, deep networks trained via supervised learning are easy to optimize and exhibit excellent generalization. One hypothesis to explain this is that overparameterized deep networks enjoy the benefits of implicit regularization induced by stochastic gradient descent, which favors parsimonious solutions that generalize well on test inputs. It is reasonable to surmise that deep reinforcement learning (RL) methods could also benefit from this effect. In this paper, we discuss how the implicit regularization effect of SGD seen in supervised learning could in fact be harmful"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2112.04716","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2112.04716/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T03:39:22Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"Bya+qlWRuIP6gUyArcGoRMAZ3ZqWJvMHZ//rbfBtLB1LLCyqqQVksg9D4pN9XWFD/XOGqDKozFpR5Mu0y/5jCA==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-05T10:34:06.182318Z"},"content_sha256":"c65eb578ae13b5dbaf7f0e7bd3cd51a4405d5b5fa57da5099a0e093709576699","schema_version":"1.0","event_id":"sha256:c65eb578ae13b5dbaf7f0e7bd3cd51a4405d5b5fa57da5099a0e093709576699"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/D5FRKACD75NVLIHYMQLWCMKIU7/bundle.json","state_url":"https://pith.science/pith/D5FRKACD75NVLIHYMQLWCMKIU7/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/D5FRKACD75NVLIHYMQLWCMKIU7/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-05T10:34:06Z","links":{"resolver":"https://pith.science/pith/D5FRKACD75NVLIHYMQLWCMKIU7","bundle":"https://pith.science/pith/D5FRKACD75NVLIHYMQLWCMKIU7/bundle.json","state":"https://pith.science/pith/D5FRKACD75NVLIHYMQLWCMKIU7/state.json","well_known_bundle":"https://pith.science/.well-known/pith/D5FRKACD75NVLIHYMQLWCMKIU7/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2021:D5FRKACD75NVLIHYMQLWCMKIU7","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"58a8e80193b842693b53e006f637a5ac674c14d9146294156e4c5ab6e20a3507","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2021-12-09T06:01:01Z","title_canon_sha256":"49e244f0893b73e12e89e08c3aa394ed04254404461809a0400008bdc3967947"},"schema_version":"1.0","source":{"id":"2112.04716","kind":"arxiv","version":1}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2112.04716","created_at":"2026-07-05T03:39:22Z"},{"alias_kind":"arxiv_version","alias_value":"2112.04716v1","created_at":"2026-07-05T03:39:22Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2112.04716","created_at":"2026-07-05T03:39:22Z"},{"alias_kind":"pith_short_12","alias_value":"D5FRKACD75NV","created_at":"2026-07-05T03:39:22Z"},{"alias_kind":"pith_short_16","alias_value":"D5FRKACD75NVLIHY","created_at":"2026-07-05T03:39:22Z"},{"alias_kind":"pith_short_8","alias_value":"D5FRKACD","created_at":"2026-07-05T03:39:22Z"}],"graph_snapshots":[{"event_id":"sha256:c65eb578ae13b5dbaf7f0e7bd3cd51a4405d5b5fa57da5099a0e093709576699","target":"graph","created_at":"2026-07-05T03:39:22Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2112.04716/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Despite overparameterization, deep networks trained via supervised learning are easy to optimize and exhibit excellent generalization. One hypothesis to explain this is that overparameterized deep networks enjoy the benefits of implicit regularization induced by stochastic gradient descent, which favors parsimonious solutions that generalize well on test inputs. It is reasonable to surmise that deep reinforcement learning (RL) methods could also benefit from this effect. In this paper, we discuss how the implicit regularization effect of SGD seen in supervised learning could in fact be harmful","authors_text":"Aaron Courville, Aviral Kumar, George Tucker, Rishabh Agarwal, Sergey Levine, Tengyu Ma","cross_cats":[],"headline":"","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2021-12-09T06:01:01Z","title":"DR3: Value-Based Deep Reinforcement Learning Requires Explicit Regularization"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2112.04716","kind":"arxiv","version":1},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:1e3697b6c4659e2c7307b8aae0322210e0038fe768782f7e5c8d8ca429beebfb","target":"record","created_at":"2026-07-05T03:39:22Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"58a8e80193b842693b53e006f637a5ac674c14d9146294156e4c5ab6e20a3507","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2021-12-09T06:01:01Z","title_canon_sha256":"49e244f0893b73e12e89e08c3aa394ed04254404461809a0400008bdc3967947"},"schema_version":"1.0","source":{"id":"2112.04716","kind":"arxiv","version":1}},"canonical_sha256":"1f4b150043ff5b55a0f86417613148a7e83d1093456709262c7d8425384dcfff","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"1f4b150043ff5b55a0f86417613148a7e83d1093456709262c7d8425384dcfff","first_computed_at":"2026-07-05T03:39:22.894985Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T03:39:22.894985Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"J5X48pnKj1F4x4xwyyq4EgyIzQ3U575bayaUdAyb+RUEeV9TtC8Xl3SkgAU8OqQ3j+Gt5OPIkrREDFWyR6tGBA==","signature_status":"signed_v1","signed_at":"2026-07-05T03:39:22.895365Z","signed_message":"canonical_sha256_bytes"},"source_id":"2112.04716","source_kind":"arxiv","source_version":1}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:1e3697b6c4659e2c7307b8aae0322210e0038fe768782f7e5c8d8ca429beebfb","sha256:c65eb578ae13b5dbaf7f0e7bd3cd51a4405d5b5fa57da5099a0e093709576699"],"state_sha256":"e002b6f601153aaee4b0bf85523f6b397a0774446cc9f9dcaa0571ce57996d3f"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"8WW/XTEqc1zhz32BmolK4XOtXBh2ouZMO7grRbm8aeodP6lesuCm6j4ngzLi6wQuWXmwXgr4HoB7IB/pF0sPAw==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-05T10:34:06.186860Z","bundle_sha256":"e997716db8551bb89045bf6fb8bf576735c447875c1a4d6ab1276f9a6460f803"}}