{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:NGXPPFG4ZW4KUETI4CW5AGWNMR","short_pith_number":"pith:NGXPPFG4","schema_version":"1.0","canonical_sha256":"69aef794dccdb8aa1268e0add01acd64478f9ed77091c3b36ce2ff0fd1abfe4f","source":{"kind":"arxiv","id":"2211.15144","version":2},"attestation_state":"computed","paper":{"title":"Offline Q-Learning on Diverse Multi-Task Data Both Scales And Generalizes","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Aviral Kumar, George Tucker, Rishabh Agarwal, Sergey Levine, Xinyang Geng","submitted_at":"2022-11-28T08:56:42Z","abstract_excerpt":"The potential of offline reinforcement learning (RL) is that high-capacity models trained on large, heterogeneous datasets can lead to agents that generalize broadly, analogously to similar advances in vision and NLP. However, recent works argue that offline RL methods encounter unique challenges to scaling up model capacity. Drawing on the learnings from these works, we re-examine previous design choices and find that with appropriate choices: ResNets, cross-entropy based distributional backups, and feature normalization, offline Q-learning algorithms exhibit strong performance that scales wi"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2211.15144","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2022-11-28T08:56:42Z","cross_cats_sorted":[],"title_canon_sha256":"4adb8b7570a9302a477a17a8dd96b49a4f05f044a47b6a3926756212edd440fc","abstract_canon_sha256":"f45a856d8fb521491c585a1f9963f1cd4467d7622ce6f1891ee6a49b689355e4"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:02:02.428159Z","signature_b64":"jQOXAKdP2IdDvPqhcI91GNEtPTzVmZRHsaQCr21vL/18iSWHFvRGNv+/8Z/zO1pTRXcKJT61HiNX3mguvbCyAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"69aef794dccdb8aa1268e0add01acd64478f9ed77091c3b36ce2ff0fd1abfe4f","last_reissued_at":"2026-07-05T06:02:02.427665Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:02:02.427665Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Offline Q-Learning on Diverse Multi-Task Data Both Scales And Generalizes","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Aviral Kumar, George Tucker, Rishabh Agarwal, Sergey Levine, Xinyang Geng","submitted_at":"2022-11-28T08:56:42Z","abstract_excerpt":"The potential of offline reinforcement learning (RL) is that high-capacity models trained on large, heterogeneous datasets can lead to agents that generalize broadly, analogously to similar advances in vision and NLP. However, recent works argue that offline RL methods encounter unique challenges to scaling up model capacity. Drawing on the learnings from these works, we re-examine previous design choices and find that with appropriate choices: ResNets, cross-entropy based distributional backups, and feature normalization, offline Q-learning algorithms exhibit strong performance that scales wi"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2211.15144","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2211.15144/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2211.15144","created_at":"2026-07-05T06:02:02.427721+00:00"},{"alias_kind":"arxiv_version","alias_value":"2211.15144v2","created_at":"2026-07-05T06:02:02.427721+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2211.15144","created_at":"2026-07-05T06:02:02.427721+00:00"},{"alias_kind":"pith_short_12","alias_value":"NGXPPFG4ZW4K","created_at":"2026-07-05T06:02:02.427721+00:00"},{"alias_kind":"pith_short_16","alias_value":"NGXPPFG4ZW4KUETI","created_at":"2026-07-05T06:02:02.427721+00:00"},{"alias_kind":"pith_short_8","alias_value":"NGXPPFG4","created_at":"2026-07-05T06:02:02.427721+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.00416","citing_title":"Learning While Deploying: Fleet-Scale Reinforcement Learning for Generalist Robot Policies","ref_index":55,"is_internal_anchor":false},{"citing_arxiv_id":"2512.20220","citing_title":"Generalisation in Multitask Fitted Q-Iteration and Offline Q-learning","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2310.06114","citing_title":"Learning Interactive Real-World Simulators","ref_index":258,"is_internal_anchor":false},{"citing_arxiv_id":"2605.00416","citing_title":"Learning While Deploying: Fleet-Scale Reinforcement Learning for Generalist Robot Policies","ref_index":55,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01663","citing_title":"Towards Efficient and Expressive Offline RL via Flow-Anchored Noise-conditioned Q-Learning","ref_index":38,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01862","citing_title":"QHyer: Q-conditioned Hybrid Attention-mamba Transformer for Offline Goal-conditioned RL","ref_index":81,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/NGXPPFG4ZW4KUETI4CW5AGWNMR","json":"https://pith.science/pith/NGXPPFG4ZW4KUETI4CW5AGWNMR.json","graph_json":"https://pith.science/api/pith-number/NGXPPFG4ZW4KUETI4CW5AGWNMR/graph.json","events_json":"https://pith.science/api/pith-number/NGXPPFG4ZW4KUETI4CW5AGWNMR/events.json","paper":"https://pith.science/paper/NGXPPFG4"},"agent_actions":{"view_html":"https://pith.science/pith/NGXPPFG4ZW4KUETI4CW5AGWNMR","download_json":"https://pith.science/pith/NGXPPFG4ZW4KUETI4CW5AGWNMR.json","view_paper":"https://pith.science/paper/NGXPPFG4","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2211.15144&json=true","fetch_graph":"https://pith.science/api/pith-number/NGXPPFG4ZW4KUETI4CW5AGWNMR/graph.json","fetch_events":"https://pith.science/api/pith-number/NGXPPFG4ZW4KUETI4CW5AGWNMR/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/NGXPPFG4ZW4KUETI4CW5AGWNMR/action/timestamp_anchor","attest_storage":"https://pith.science/pith/NGXPPFG4ZW4KUETI4CW5AGWNMR/action/storage_attestation","attest_author":"https://pith.science/pith/NGXPPFG4ZW4KUETI4CW5AGWNMR/action/author_attestation","sign_citation":"https://pith.science/pith/NGXPPFG4ZW4KUETI4CW5AGWNMR/action/citation_signature","submit_replication":"https://pith.science/pith/NGXPPFG4ZW4KUETI4CW5AGWNMR/action/replication_record"}},"created_at":"2026-07-05T06:02:02.427721+00:00","updated_at":"2026-07-05T06:02:02.427721+00:00"}