{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2021:HKFZGHTCRTSIMPQRAM2MWFBKB7","short_pith_number":"pith:HKFZGHTC","canonical_record":{"source":{"id":"2103.04947","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2021-03-08T18:06:44Z","cross_cats_sorted":["cs.AI","math.OC","stat.ML"],"title_canon_sha256":"25c50c3a9841239248c0889fc479945bd2a41987986da724abcb9f75c128c700","abstract_canon_sha256":"c7d1869d200ddcb7d42229abef2499aac1a93ea214fdf439eeef7f7aafa144d7"},"schema_version":"1.0"},"canonical_sha256":"3a8b931e628ce4863e110334cb142a0ffce16fa7a6c9aead0f7c0b8230ff04c8","source":{"kind":"arxiv","id":"2103.04947","version":1},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2103.04947","created_at":"2026-07-05T02:21:10Z"},{"alias_kind":"arxiv_version","alias_value":"2103.04947v1","created_at":"2026-07-05T02:21:10Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2103.04947","created_at":"2026-07-05T02:21:10Z"},{"alias_kind":"pith_short_12","alias_value":"HKFZGHTCRTSI","created_at":"2026-07-05T02:21:10Z"},{"alias_kind":"pith_short_16","alias_value":"HKFZGHTCRTSIMPQR","created_at":"2026-07-05T02:21:10Z"},{"alias_kind":"pith_short_8","alias_value":"HKFZGHTC","created_at":"2026-07-05T02:21:10Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2021:HKFZGHTCRTSIMPQRAM2MWFBKB7","target":"record","payload":{"canonical_record":{"source":{"id":"2103.04947","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2021-03-08T18:06:44Z","cross_cats_sorted":["cs.AI","math.OC","stat.ML"],"title_canon_sha256":"25c50c3a9841239248c0889fc479945bd2a41987986da724abcb9f75c128c700","abstract_canon_sha256":"c7d1869d200ddcb7d42229abef2499aac1a93ea214fdf439eeef7f7aafa144d7"},"schema_version":"1.0"},"canonical_sha256":"3a8b931e628ce4863e110334cb142a0ffce16fa7a6c9aead0f7c0b8230ff04c8","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T02:21:10.631242Z","signature_b64":"qXBUOMWsTIekNaaVDz69WCXg2c9LZwmhH9zfz0oCeYBKwW/nSn2n2SCV+gW0qhRBySolloJSxIIX49iJZGB0DQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"3a8b931e628ce4863e110334cb142a0ffce16fa7a6c9aead0f7c0b8230ff04c8","last_reissued_at":"2026-07-05T02:21:10.630720Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T02:21:10.630720Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2103.04947","source_version":1,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T02:21:10Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"zJkeSYvgi7eNwRslt6d2nus2G7BnjZUvUaVdYj0sajRMrBLXyIqiGC+D45/yvFBtzTFDpHRm1HXJ1ad+5s/HAA==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-10T22:35:04.863534Z"},"content_sha256":"51c5ee68195da45b2e686df333fe582b7e56d68db2c757766caa531b957e8fd3","schema_version":"1.0","event_id":"sha256:51c5ee68195da45b2e686df333fe582b7e56d68db2c757766caa531b957e8fd3"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2021:HKFZGHTCRTSIMPQRAM2MWFBKB7","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Instabilities of Offline RL with Pre-Trained Neural Representation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","math.OC","stat.ML"],"primary_cat":"cs.LG","authors_text":"Ruosong Wang, Ruslan Salakhutdinov, Sham M. Kakade, Yifan Wu","submitted_at":"2021-03-08T18:06:44Z","abstract_excerpt":"In offline reinforcement learning (RL), we seek to utilize offline data to evaluate (or learn) policies in scenarios where the data are collected from a distribution that substantially differs from that of the target policy to be evaluated. Recent theoretical advances have shown that such sample-efficient offline RL is indeed possible provided certain strong representational conditions hold, else there are lower bounds exhibiting exponential error amplification (in the problem horizon) unless the data collection distribution has only a mild distribution shift relative to the target policy. Thi"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2103.04947","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2103.04947/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T02:21:10Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"VtBHVwo+NzVw6G1lWPudT23YbzlKJivyxaIvN6YcGcxVy+Bzp6/14gbR3XnIkihiTXSz/bmBx+q8hvAuGQwCCw==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-10T22:35:04.864178Z"},"content_sha256":"43289649c03992a393512e8faa2edc7011cf8948dd229b2cdc86ef5a4236230e","schema_version":"1.0","event_id":"sha256:43289649c03992a393512e8faa2edc7011cf8948dd229b2cdc86ef5a4236230e"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/HKFZGHTCRTSIMPQRAM2MWFBKB7/bundle.json","state_url":"https://pith.science/pith/HKFZGHTCRTSIMPQRAM2MWFBKB7/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/HKFZGHTCRTSIMPQRAM2MWFBKB7/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-10T22:35:04Z","links":{"resolver":"https://pith.science/pith/HKFZGHTCRTSIMPQRAM2MWFBKB7","bundle":"https://pith.science/pith/HKFZGHTCRTSIMPQRAM2MWFBKB7/bundle.json","state":"https://pith.science/pith/HKFZGHTCRTSIMPQRAM2MWFBKB7/state.json","well_known_bundle":"https://pith.science/.well-known/pith/HKFZGHTCRTSIMPQRAM2MWFBKB7/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2021:HKFZGHTCRTSIMPQRAM2MWFBKB7","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"c7d1869d200ddcb7d42229abef2499aac1a93ea214fdf439eeef7f7aafa144d7","cross_cats_sorted":["cs.AI","math.OC","stat.ML"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2021-03-08T18:06:44Z","title_canon_sha256":"25c50c3a9841239248c0889fc479945bd2a41987986da724abcb9f75c128c700"},"schema_version":"1.0","source":{"id":"2103.04947","kind":"arxiv","version":1}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2103.04947","created_at":"2026-07-05T02:21:10Z"},{"alias_kind":"arxiv_version","alias_value":"2103.04947v1","created_at":"2026-07-05T02:21:10Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2103.04947","created_at":"2026-07-05T02:21:10Z"},{"alias_kind":"pith_short_12","alias_value":"HKFZGHTCRTSI","created_at":"2026-07-05T02:21:10Z"},{"alias_kind":"pith_short_16","alias_value":"HKFZGHTCRTSIMPQR","created_at":"2026-07-05T02:21:10Z"},{"alias_kind":"pith_short_8","alias_value":"HKFZGHTC","created_at":"2026-07-05T02:21:10Z"}],"graph_snapshots":[{"event_id":"sha256:43289649c03992a393512e8faa2edc7011cf8948dd229b2cdc86ef5a4236230e","target":"graph","created_at":"2026-07-05T02:21:10Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2103.04947/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"In offline reinforcement learning (RL), we seek to utilize offline data to evaluate (or learn) policies in scenarios where the data are collected from a distribution that substantially differs from that of the target policy to be evaluated. Recent theoretical advances have shown that such sample-efficient offline RL is indeed possible provided certain strong representational conditions hold, else there are lower bounds exhibiting exponential error amplification (in the problem horizon) unless the data collection distribution has only a mild distribution shift relative to the target policy. Thi","authors_text":"Ruosong Wang, Ruslan Salakhutdinov, Sham M. Kakade, Yifan Wu","cross_cats":["cs.AI","math.OC","stat.ML"],"headline":"","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2021-03-08T18:06:44Z","title":"Instabilities of Offline RL with Pre-Trained Neural Representation"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2103.04947","kind":"arxiv","version":1},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:51c5ee68195da45b2e686df333fe582b7e56d68db2c757766caa531b957e8fd3","target":"record","created_at":"2026-07-05T02:21:10Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"c7d1869d200ddcb7d42229abef2499aac1a93ea214fdf439eeef7f7aafa144d7","cross_cats_sorted":["cs.AI","math.OC","stat.ML"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2021-03-08T18:06:44Z","title_canon_sha256":"25c50c3a9841239248c0889fc479945bd2a41987986da724abcb9f75c128c700"},"schema_version":"1.0","source":{"id":"2103.04947","kind":"arxiv","version":1}},"canonical_sha256":"3a8b931e628ce4863e110334cb142a0ffce16fa7a6c9aead0f7c0b8230ff04c8","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"3a8b931e628ce4863e110334cb142a0ffce16fa7a6c9aead0f7c0b8230ff04c8","first_computed_at":"2026-07-05T02:21:10.630720Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T02:21:10.630720Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"qXBUOMWsTIekNaaVDz69WCXg2c9LZwmhH9zfz0oCeYBKwW/nSn2n2SCV+gW0qhRBySolloJSxIIX49iJZGB0DQ==","signature_status":"signed_v1","signed_at":"2026-07-05T02:21:10.631242Z","signed_message":"canonical_sha256_bytes"},"source_id":"2103.04947","source_kind":"arxiv","source_version":1}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:51c5ee68195da45b2e686df333fe582b7e56d68db2c757766caa531b957e8fd3","sha256:43289649c03992a393512e8faa2edc7011cf8948dd229b2cdc86ef5a4236230e"],"state_sha256":"3a91c118227b66e9c0e155754de38288569a790ce50b8cf18b83fbacad5c43d9"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"zcZ9VqaUbm8+I3pDgXQssQH3Z8Vz9T6IYH1e2L31krCDidLAE6glO6lnONxKXFPZq98eWEopogEenww2AUU+AA==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-10T22:35:04.869727Z","bundle_sha256":"9a725eb9bcba427cf874c825d12df0a96d2d487f49a66b8e4e9f1f54d910468c"}}