{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:NFOD7ABY656FVQKYLWYKFO3ZBM","short_pith_number":"pith:NFOD7ABY","schema_version":"1.0","canonical_sha256":"695c3f8038f77c5ac1585db0a2bb790b23bd2dbce0196eeb846a5adaa851e9ef","source":{"kind":"arxiv","id":"2207.02099","version":2},"attestation_state":"computed","paper":{"title":"An Empirical Study of Implicit Regularization in Deep Offline RL","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Arnaud Doucet, Caglar Gulcehre, Georg Ostrovski, Jakub Sygnowski, Matt Hoffman, Mehrdad Farajtabar, Razvan Pascanu, Srivatsan Srinivasan","submitted_at":"2022-07-05T15:07:31Z","abstract_excerpt":"Deep neural networks are the most commonly used function approximators in offline reinforcement learning. Prior works have shown that neural nets trained with TD-learning and gradient descent can exhibit implicit regularization that can be characterized by under-parameterization of these networks. Specifically, the rank of the penultimate feature layer, also called \\textit{effective rank}, has been observed to drastically collapse during the training. In turn, this collapse has been argued to reduce the model's ability to further adapt in later stages of learning, leading to the diminished fin"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2207.02099","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2022-07-05T15:07:31Z","cross_cats_sorted":[],"title_canon_sha256":"e39ba6e205f6ebc271bcece69160cd1173f8ba1555ebc6985d3e661db04511d1","abstract_canon_sha256":"7ee13df861d9581008d08449e410ed27b733ac943a737f79b461244fa5d02bd3"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T04:38:16.260211Z","signature_b64":"bGFwJNlF2nrG3Dq3yYY8MQ0rSdPr+ibCA/o0P1AyavSudffD7uvXiDjyVbxqZmW5Ixd/sGH4tpo/1uE5SO47Cg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"695c3f8038f77c5ac1585db0a2bb790b23bd2dbce0196eeb846a5adaa851e9ef","last_reissued_at":"2026-07-05T04:38:16.259757Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T04:38:16.259757Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"An Empirical Study of Implicit Regularization in Deep Offline RL","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Arnaud Doucet, Caglar Gulcehre, Georg Ostrovski, Jakub Sygnowski, Matt Hoffman, Mehrdad Farajtabar, Razvan Pascanu, Srivatsan Srinivasan","submitted_at":"2022-07-05T15:07:31Z","abstract_excerpt":"Deep neural networks are the most commonly used function approximators in offline reinforcement learning. Prior works have shown that neural nets trained with TD-learning and gradient descent can exhibit implicit regularization that can be characterized by under-parameterization of these networks. Specifically, the rank of the penultimate feature layer, also called \\textit{effective rank}, has been observed to drastically collapse during the training. In turn, this collapse has been argued to reduce the model's ability to further adapt in later stages of learning, leading to the diminished fin"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2207.02099","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2207.02099/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2207.02099","created_at":"2026-07-05T04:38:16.259810+00:00"},{"alias_kind":"arxiv_version","alias_value":"2207.02099v2","created_at":"2026-07-05T04:38:16.259810+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2207.02099","created_at":"2026-07-05T04:38:16.259810+00:00"},{"alias_kind":"pith_short_12","alias_value":"NFOD7ABY656F","created_at":"2026-07-05T04:38:16.259810+00:00"},{"alias_kind":"pith_short_16","alias_value":"NFOD7ABY656FVQKY","created_at":"2026-07-05T04:38:16.259810+00:00"},{"alias_kind":"pith_short_8","alias_value":"NFOD7ABY","created_at":"2026-07-05T04:38:16.259810+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2604.13966","citing_title":"Provably Efficient Offline-to-Online Value Adaptation with General Function Approximation","ref_index":4,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/NFOD7ABY656FVQKYLWYKFO3ZBM","json":"https://pith.science/pith/NFOD7ABY656FVQKYLWYKFO3ZBM.json","graph_json":"https://pith.science/api/pith-number/NFOD7ABY656FVQKYLWYKFO3ZBM/graph.json","events_json":"https://pith.science/api/pith-number/NFOD7ABY656FVQKYLWYKFO3ZBM/events.json","paper":"https://pith.science/paper/NFOD7ABY"},"agent_actions":{"view_html":"https://pith.science/pith/NFOD7ABY656FVQKYLWYKFO3ZBM","download_json":"https://pith.science/pith/NFOD7ABY656FVQKYLWYKFO3ZBM.json","view_paper":"https://pith.science/paper/NFOD7ABY","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2207.02099&json=true","fetch_graph":"https://pith.science/api/pith-number/NFOD7ABY656FVQKYLWYKFO3ZBM/graph.json","fetch_events":"https://pith.science/api/pith-number/NFOD7ABY656FVQKYLWYKFO3ZBM/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/NFOD7ABY656FVQKYLWYKFO3ZBM/action/timestamp_anchor","attest_storage":"https://pith.science/pith/NFOD7ABY656FVQKYLWYKFO3ZBM/action/storage_attestation","attest_author":"https://pith.science/pith/NFOD7ABY656FVQKYLWYKFO3ZBM/action/author_attestation","sign_citation":"https://pith.science/pith/NFOD7ABY656FVQKYLWYKFO3ZBM/action/citation_signature","submit_replication":"https://pith.science/pith/NFOD7ABY656FVQKYLWYKFO3ZBM/action/replication_record"}},"created_at":"2026-07-05T04:38:16.259810+00:00","updated_at":"2026-07-05T04:38:16.259810+00:00"}