{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2022:NGXPPFG4ZW4KUETI4CW5AGWNMR","short_pith_number":"pith:NGXPPFG4","canonical_record":{"source":{"id":"2211.15144","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2022-11-28T08:56:42Z","cross_cats_sorted":[],"title_canon_sha256":"4adb8b7570a9302a477a17a8dd96b49a4f05f044a47b6a3926756212edd440fc","abstract_canon_sha256":"f45a856d8fb521491c585a1f9963f1cd4467d7622ce6f1891ee6a49b689355e4"},"schema_version":"1.0"},"canonical_sha256":"69aef794dccdb8aa1268e0add01acd64478f9ed77091c3b36ce2ff0fd1abfe4f","source":{"kind":"arxiv","id":"2211.15144","version":2},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2211.15144","created_at":"2026-07-05T06:02:02Z"},{"alias_kind":"arxiv_version","alias_value":"2211.15144v2","created_at":"2026-07-05T06:02:02Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2211.15144","created_at":"2026-07-05T06:02:02Z"},{"alias_kind":"pith_short_12","alias_value":"NGXPPFG4ZW4K","created_at":"2026-07-05T06:02:02Z"},{"alias_kind":"pith_short_16","alias_value":"NGXPPFG4ZW4KUETI","created_at":"2026-07-05T06:02:02Z"},{"alias_kind":"pith_short_8","alias_value":"NGXPPFG4","created_at":"2026-07-05T06:02:02Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2022:NGXPPFG4ZW4KUETI4CW5AGWNMR","target":"record","payload":{"canonical_record":{"source":{"id":"2211.15144","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2022-11-28T08:56:42Z","cross_cats_sorted":[],"title_canon_sha256":"4adb8b7570a9302a477a17a8dd96b49a4f05f044a47b6a3926756212edd440fc","abstract_canon_sha256":"f45a856d8fb521491c585a1f9963f1cd4467d7622ce6f1891ee6a49b689355e4"},"schema_version":"1.0"},"canonical_sha256":"69aef794dccdb8aa1268e0add01acd64478f9ed77091c3b36ce2ff0fd1abfe4f","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:02:02.428159Z","signature_b64":"jQOXAKdP2IdDvPqhcI91GNEtPTzVmZRHsaQCr21vL/18iSWHFvRGNv+/8Z/zO1pTRXcKJT61HiNX3mguvbCyAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"69aef794dccdb8aa1268e0add01acd64478f9ed77091c3b36ce2ff0fd1abfe4f","last_reissued_at":"2026-07-05T06:02:02.427665Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:02:02.427665Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2211.15144","source_version":2,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T06:02:02Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"++iukTOgRna+fTHpTVBp9I/WQTii/ToevKHQKahzsHxli+c4/+J9QbFpJ6A4x21O/oVeCkSGg4SZaABl9ufQCw==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-01T10:25:31.134803Z"},"content_sha256":"c790513d0e7ad58d6986f4792fba9056e8be55111511600a4922884faa39542d","schema_version":"1.0","event_id":"sha256:c790513d0e7ad58d6986f4792fba9056e8be55111511600a4922884faa39542d"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2022:NGXPPFG4ZW4KUETI4CW5AGWNMR","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Offline Q-Learning on Diverse Multi-Task Data Both Scales And Generalizes","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Aviral Kumar, George Tucker, Rishabh Agarwal, Sergey Levine, Xinyang Geng","submitted_at":"2022-11-28T08:56:42Z","abstract_excerpt":"The potential of offline reinforcement learning (RL) is that high-capacity models trained on large, heterogeneous datasets can lead to agents that generalize broadly, analogously to similar advances in vision and NLP. However, recent works argue that offline RL methods encounter unique challenges to scaling up model capacity. Drawing on the learnings from these works, we re-examine previous design choices and find that with appropriate choices: ResNets, cross-entropy based distributional backups, and feature normalization, offline Q-learning algorithms exhibit strong performance that scales wi"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2211.15144","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2211.15144/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T06:02:02Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"U5EAMS0oziEpPg6IXIx8AJWSW8950BRBmD9wWHgmlPYfB8pTnbKPjwqZ3zcwui/AxBvs9cPVdnMJFHkoxjK4BQ==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-01T10:25:31.135179Z"},"content_sha256":"8fd03c2e76059034a39d0d4f352493f761018d7726a39e285fa21195bc018d68","schema_version":"1.0","event_id":"sha256:8fd03c2e76059034a39d0d4f352493f761018d7726a39e285fa21195bc018d68"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/NGXPPFG4ZW4KUETI4CW5AGWNMR/bundle.json","state_url":"https://pith.science/pith/NGXPPFG4ZW4KUETI4CW5AGWNMR/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/NGXPPFG4ZW4KUETI4CW5AGWNMR/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-01T10:25:31Z","links":{"resolver":"https://pith.science/pith/NGXPPFG4ZW4KUETI4CW5AGWNMR","bundle":"https://pith.science/pith/NGXPPFG4ZW4KUETI4CW5AGWNMR/bundle.json","state":"https://pith.science/pith/NGXPPFG4ZW4KUETI4CW5AGWNMR/state.json","well_known_bundle":"https://pith.science/.well-known/pith/NGXPPFG4ZW4KUETI4CW5AGWNMR/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2022:NGXPPFG4ZW4KUETI4CW5AGWNMR","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"f45a856d8fb521491c585a1f9963f1cd4467d7622ce6f1891ee6a49b689355e4","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2022-11-28T08:56:42Z","title_canon_sha256":"4adb8b7570a9302a477a17a8dd96b49a4f05f044a47b6a3926756212edd440fc"},"schema_version":"1.0","source":{"id":"2211.15144","kind":"arxiv","version":2}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2211.15144","created_at":"2026-07-05T06:02:02Z"},{"alias_kind":"arxiv_version","alias_value":"2211.15144v2","created_at":"2026-07-05T06:02:02Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2211.15144","created_at":"2026-07-05T06:02:02Z"},{"alias_kind":"pith_short_12","alias_value":"NGXPPFG4ZW4K","created_at":"2026-07-05T06:02:02Z"},{"alias_kind":"pith_short_16","alias_value":"NGXPPFG4ZW4KUETI","created_at":"2026-07-05T06:02:02Z"},{"alias_kind":"pith_short_8","alias_value":"NGXPPFG4","created_at":"2026-07-05T06:02:02Z"}],"graph_snapshots":[{"event_id":"sha256:8fd03c2e76059034a39d0d4f352493f761018d7726a39e285fa21195bc018d68","target":"graph","created_at":"2026-07-05T06:02:02Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2211.15144/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"The potential of offline reinforcement learning (RL) is that high-capacity models trained on large, heterogeneous datasets can lead to agents that generalize broadly, analogously to similar advances in vision and NLP. However, recent works argue that offline RL methods encounter unique challenges to scaling up model capacity. Drawing on the learnings from these works, we re-examine previous design choices and find that with appropriate choices: ResNets, cross-entropy based distributional backups, and feature normalization, offline Q-learning algorithms exhibit strong performance that scales wi","authors_text":"Aviral Kumar, George Tucker, Rishabh Agarwal, Sergey Levine, Xinyang Geng","cross_cats":[],"headline":"","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2022-11-28T08:56:42Z","title":"Offline Q-Learning on Diverse Multi-Task Data Both Scales And Generalizes"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2211.15144","kind":"arxiv","version":2},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:c790513d0e7ad58d6986f4792fba9056e8be55111511600a4922884faa39542d","target":"record","created_at":"2026-07-05T06:02:02Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"f45a856d8fb521491c585a1f9963f1cd4467d7622ce6f1891ee6a49b689355e4","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2022-11-28T08:56:42Z","title_canon_sha256":"4adb8b7570a9302a477a17a8dd96b49a4f05f044a47b6a3926756212edd440fc"},"schema_version":"1.0","source":{"id":"2211.15144","kind":"arxiv","version":2}},"canonical_sha256":"69aef794dccdb8aa1268e0add01acd64478f9ed77091c3b36ce2ff0fd1abfe4f","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"69aef794dccdb8aa1268e0add01acd64478f9ed77091c3b36ce2ff0fd1abfe4f","first_computed_at":"2026-07-05T06:02:02.427665Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T06:02:02.427665Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"jQOXAKdP2IdDvPqhcI91GNEtPTzVmZRHsaQCr21vL/18iSWHFvRGNv+/8Z/zO1pTRXcKJT61HiNX3mguvbCyAw==","signature_status":"signed_v1","signed_at":"2026-07-05T06:02:02.428159Z","signed_message":"canonical_sha256_bytes"},"source_id":"2211.15144","source_kind":"arxiv","source_version":2}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:c790513d0e7ad58d6986f4792fba9056e8be55111511600a4922884faa39542d","sha256:8fd03c2e76059034a39d0d4f352493f761018d7726a39e285fa21195bc018d68"],"state_sha256":"a9e11c8027ef41d5fbc5eccba5e5750463391aa40995ceb0404eb308a9a55f41"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"OMqA8fOoxApeqw5qIo367sAKOntOQUCD1cVNqA9UZVxVLjX1dJOrM3Jw+YoVM7+KrgcGLIZRR1dDUWiWyhXkCg==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-01T10:25:31.137995Z","bundle_sha256":"f10512ebc7874d91cea2e6f11ceff2961c412ab2988b802f183e9ab7db61e4c0"}}