{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2021:33TJGYJKQP7NB7C2S2U6XRZGB3","short_pith_number":"pith:33TJGYJK","canonical_record":{"source":{"id":"2102.05815","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2021-02-11T02:38:12Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"cd0c35db0f64640d49ce6665c92f93c407c15f8fed6918c1a09b48b0b5b6a610","abstract_canon_sha256":"e3035a12d1df82750c9b59f6ee8ec4a68e34e18274fd9e28e74857eaa56e4cb4"},"schema_version":"1.0"},"canonical_sha256":"dee693612a83fed0fc5a96a9ebc7260ee30e4d1606966c9c834180690633633c","source":{"kind":"arxiv","id":"2102.05815","version":1},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2102.05815","created_at":"2026-07-05T02:14:34Z"},{"alias_kind":"arxiv_version","alias_value":"2102.05815v1","created_at":"2026-07-05T02:14:34Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2102.05815","created_at":"2026-07-05T02:14:34Z"},{"alias_kind":"pith_short_12","alias_value":"33TJGYJKQP7N","created_at":"2026-07-05T02:14:34Z"},{"alias_kind":"pith_short_16","alias_value":"33TJGYJKQP7NB7C2","created_at":"2026-07-05T02:14:34Z"},{"alias_kind":"pith_short_8","alias_value":"33TJGYJK","created_at":"2026-07-05T02:14:34Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2021:33TJGYJKQP7NB7C2S2U6XRZGB3","target":"record","payload":{"canonical_record":{"source":{"id":"2102.05815","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2021-02-11T02:38:12Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"cd0c35db0f64640d49ce6665c92f93c407c15f8fed6918c1a09b48b0b5b6a610","abstract_canon_sha256":"e3035a12d1df82750c9b59f6ee8ec4a68e34e18274fd9e28e74857eaa56e4cb4"},"schema_version":"1.0"},"canonical_sha256":"dee693612a83fed0fc5a96a9ebc7260ee30e4d1606966c9c834180690633633c","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T02:14:34.853707Z","signature_b64":"IicYL9AuGFC+vgB7YIiaZePSYw0lM2ck2ZmZFxrj5JJoc8Qb5eWx55xq5qvb27XM4ooHplwVrrC1PW96RVDuBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"dee693612a83fed0fc5a96a9ebc7260ee30e4d1606966c9c834180690633633c","last_reissued_at":"2026-07-05T02:14:34.853228Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T02:14:34.853228Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2102.05815","source_version":1,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T02:14:34Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"b9wteB56apd45qREC+zi1WIsADtEnWy2LPSBvtDyE3VFZRXBVXRTb08Ox+EGyDs612BOp8KJ6pPQypTwqvXkBQ==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-18T09:10:56.128181Z"},"content_sha256":"dc03d0dca6b18832b8bf62cae35b27a5d51dee75bed859f80f548c5004748ad0","schema_version":"1.0","event_id":"sha256:dc03d0dca6b18832b8bf62cae35b27a5d51dee75bed859f80f548c5004748ad0"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2021:33TJGYJKQP7NB7C2S2U6XRZGB3","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Representation Matters: Offline Pretraining for Sequential Decision Making","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Mengjiao Yang, Ofir Nachum","submitted_at":"2021-02-11T02:38:12Z","abstract_excerpt":"The recent success of supervised learning methods on ever larger offline datasets has spurred interest in the reinforcement learning (RL) field to investigate whether the same paradigms can be translated to RL algorithms. This research area, known as offline RL, has largely focused on offline policy optimization, aiming to find a return-maximizing policy exclusively from offline data. In this paper, we consider a slightly different approach to incorporating offline data into sequential decision-making. We aim to answer the question, what unsupervised objectives applied to offline datasets are "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2102.05815","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2102.05815/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T02:14:34Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"s2zXnCTreMyyzR3oeG6k8yxPiF2hW5kSTusvwH+IuLGaxmrNSLXierHNn14x6iQt3IC5R3vcEITfohBRjiNEDg==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-18T09:10:56.128738Z"},"content_sha256":"084b944361043f809766f920b77a04688b41b909f602a6427f2178bfaf45abe3","schema_version":"1.0","event_id":"sha256:084b944361043f809766f920b77a04688b41b909f602a6427f2178bfaf45abe3"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/33TJGYJKQP7NB7C2S2U6XRZGB3/bundle.json","state_url":"https://pith.science/pith/33TJGYJKQP7NB7C2S2U6XRZGB3/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/33TJGYJKQP7NB7C2S2U6XRZGB3/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-18T09:10:56Z","links":{"resolver":"https://pith.science/pith/33TJGYJKQP7NB7C2S2U6XRZGB3","bundle":"https://pith.science/pith/33TJGYJKQP7NB7C2S2U6XRZGB3/bundle.json","state":"https://pith.science/pith/33TJGYJKQP7NB7C2S2U6XRZGB3/state.json","well_known_bundle":"https://pith.science/.well-known/pith/33TJGYJKQP7NB7C2S2U6XRZGB3/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2021:33TJGYJKQP7NB7C2S2U6XRZGB3","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"e3035a12d1df82750c9b59f6ee8ec4a68e34e18274fd9e28e74857eaa56e4cb4","cross_cats_sorted":["cs.AI"],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2021-02-11T02:38:12Z","title_canon_sha256":"cd0c35db0f64640d49ce6665c92f93c407c15f8fed6918c1a09b48b0b5b6a610"},"schema_version":"1.0","source":{"id":"2102.05815","kind":"arxiv","version":1}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2102.05815","created_at":"2026-07-05T02:14:34Z"},{"alias_kind":"arxiv_version","alias_value":"2102.05815v1","created_at":"2026-07-05T02:14:34Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2102.05815","created_at":"2026-07-05T02:14:34Z"},{"alias_kind":"pith_short_12","alias_value":"33TJGYJKQP7N","created_at":"2026-07-05T02:14:34Z"},{"alias_kind":"pith_short_16","alias_value":"33TJGYJKQP7NB7C2","created_at":"2026-07-05T02:14:34Z"},{"alias_kind":"pith_short_8","alias_value":"33TJGYJK","created_at":"2026-07-05T02:14:34Z"}],"graph_snapshots":[{"event_id":"sha256:084b944361043f809766f920b77a04688b41b909f602a6427f2178bfaf45abe3","target":"graph","created_at":"2026-07-05T02:14:34Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2102.05815/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"The recent success of supervised learning methods on ever larger offline datasets has spurred interest in the reinforcement learning (RL) field to investigate whether the same paradigms can be translated to RL algorithms. This research area, known as offline RL, has largely focused on offline policy optimization, aiming to find a return-maximizing policy exclusively from offline data. In this paper, we consider a slightly different approach to incorporating offline data into sequential decision-making. We aim to answer the question, what unsupervised objectives applied to offline datasets are ","authors_text":"Mengjiao Yang, Ofir Nachum","cross_cats":["cs.AI"],"headline":"","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2021-02-11T02:38:12Z","title":"Representation Matters: Offline Pretraining for Sequential Decision Making"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2102.05815","kind":"arxiv","version":1},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:dc03d0dca6b18832b8bf62cae35b27a5d51dee75bed859f80f548c5004748ad0","target":"record","created_at":"2026-07-05T02:14:34Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"e3035a12d1df82750c9b59f6ee8ec4a68e34e18274fd9e28e74857eaa56e4cb4","cross_cats_sorted":["cs.AI"],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2021-02-11T02:38:12Z","title_canon_sha256":"cd0c35db0f64640d49ce6665c92f93c407c15f8fed6918c1a09b48b0b5b6a610"},"schema_version":"1.0","source":{"id":"2102.05815","kind":"arxiv","version":1}},"canonical_sha256":"dee693612a83fed0fc5a96a9ebc7260ee30e4d1606966c9c834180690633633c","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"dee693612a83fed0fc5a96a9ebc7260ee30e4d1606966c9c834180690633633c","first_computed_at":"2026-07-05T02:14:34.853228Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T02:14:34.853228Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"IicYL9AuGFC+vgB7YIiaZePSYw0lM2ck2ZmZFxrj5JJoc8Qb5eWx55xq5qvb27XM4ooHplwVrrC1PW96RVDuBw==","signature_status":"signed_v1","signed_at":"2026-07-05T02:14:34.853707Z","signed_message":"canonical_sha256_bytes"},"source_id":"2102.05815","source_kind":"arxiv","source_version":1}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:dc03d0dca6b18832b8bf62cae35b27a5d51dee75bed859f80f548c5004748ad0","sha256:084b944361043f809766f920b77a04688b41b909f602a6427f2178bfaf45abe3"],"state_sha256":"a7c3b2e6412776f5974f583322a19d12cda14c2f3612ad4eea2c66738f41a631"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"TEpzNCPP3izY0aHB3X00xHiUGAHxyj/1mPlMIaZJ65M8UuKyiipcYNb7amppQUfJQmbb+Q/FoBalYNRbpMdzDQ==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-18T09:10:56.133778Z","bundle_sha256":"107c3e1bc53b2679419bc047d0f5478f5fea594897c4d72716f06380fba0e6d3"}}