{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2020:WL7EVLFBMZSZMQN7YX5APTD5D4","short_pith_number":"pith:WL7EVLFB","canonical_record":{"source":{"id":"2002.11089","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2020-02-25T18:36:31Z","cross_cats_sorted":["cs.AI","cs.RO","stat.ML"],"title_canon_sha256":"5c6edeba7e7fe6323d2bf5f7a3bcee6214be887c2d052c1d773e3483466d6d93","abstract_canon_sha256":"da19a77e56762f2950411c9973d4768b2c15255795c9a630e6127063835b0b4b"},"schema_version":"1.0"},"canonical_sha256":"b2fe4aaca166659641bfc5fa07cc7d1f2c0894fcf403f34fb352de6999d17dde","source":{"kind":"arxiv","id":"2002.11089","version":1},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2002.11089","created_at":"2026-07-05T00:43:51Z"},{"alias_kind":"arxiv_version","alias_value":"2002.11089v1","created_at":"2026-07-05T00:43:51Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2002.11089","created_at":"2026-07-05T00:43:51Z"},{"alias_kind":"pith_short_12","alias_value":"WL7EVLFBMZSZ","created_at":"2026-07-05T00:43:51Z"},{"alias_kind":"pith_short_16","alias_value":"WL7EVLFBMZSZMQN7","created_at":"2026-07-05T00:43:51Z"},{"alias_kind":"pith_short_8","alias_value":"WL7EVLFB","created_at":"2026-07-05T00:43:51Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2020:WL7EVLFBMZSZMQN7YX5APTD5D4","target":"record","payload":{"canonical_record":{"source":{"id":"2002.11089","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2020-02-25T18:36:31Z","cross_cats_sorted":["cs.AI","cs.RO","stat.ML"],"title_canon_sha256":"5c6edeba7e7fe6323d2bf5f7a3bcee6214be887c2d052c1d773e3483466d6d93","abstract_canon_sha256":"da19a77e56762f2950411c9973d4768b2c15255795c9a630e6127063835b0b4b"},"schema_version":"1.0"},"canonical_sha256":"b2fe4aaca166659641bfc5fa07cc7d1f2c0894fcf403f34fb352de6999d17dde","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T00:43:51.491416Z","signature_b64":"hA6dyVIaiWeZJmlx2O2sEtzM+BT/ty0+EHJ3rQwcZZPqf/3xzaZHTqwyNLEJB/fiuZv3t75zqahhxeOjJQDDCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b2fe4aaca166659641bfc5fa07cc7d1f2c0894fcf403f34fb352de6999d17dde","last_reissued_at":"2026-07-05T00:43:51.490990Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T00:43:51.490990Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2002.11089","source_version":1,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T00:43:51Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"AE3i5jvwM1wqPydMafWQBgYhMNo+BsXstMzlZncqhblD+XUtUa6D3jNeV19oVgFQ8zLUcfqnKcAEBeLQnKoSCA==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-09T19:34:11.795811Z"},"content_sha256":"40da3af8b59a0ba224b8df1c0dfd27749a9b43f40dbe3ce597356e3687b79a2a","schema_version":"1.0","event_id":"sha256:40da3af8b59a0ba224b8df1c0dfd27749a9b43f40dbe3ce597356e3687b79a2a"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2020:WL7EVLFBMZSZMQN7YX5APTD5D4","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Rewriting History with Inverse RL: Hindsight Inference for Policy Improvement","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.RO","stat.ML"],"primary_cat":"cs.LG","authors_text":"Benjamin Eysenbach, Ruslan Salakhutdinov, Sergey Levine, Xinyang Geng","submitted_at":"2020-02-25T18:36:31Z","abstract_excerpt":"Multi-task reinforcement learning (RL) aims to simultaneously learn policies for solving many tasks. Several prior works have found that relabeling past experience with different reward functions can improve sample efficiency. Relabeling methods typically ask: if, in hindsight, we assume that our experience was optimal for some task, for what task was it optimal? In this paper, we show that hindsight relabeling is inverse RL, an observation that suggests that we can use inverse RL in tandem for RL algorithms to efficiently solve many tasks. We use this idea to generalize goal-relabeling techni"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2002.11089","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2002.11089/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T00:43:51Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"0MM4EVh5QXCc1QIRTy9GITx/2jc/wNNzbWovUHaPSG2OhjjwHQMgEe0Nlz/3uuSA5LBrJQTGhFtJzsY6EkASAw==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-09T19:34:11.796333Z"},"content_sha256":"c50e442da41987be6d0cc3f9435711b148a71ef08e909e7366e1593308cca2ca","schema_version":"1.0","event_id":"sha256:c50e442da41987be6d0cc3f9435711b148a71ef08e909e7366e1593308cca2ca"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/WL7EVLFBMZSZMQN7YX5APTD5D4/bundle.json","state_url":"https://pith.science/pith/WL7EVLFBMZSZMQN7YX5APTD5D4/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/WL7EVLFBMZSZMQN7YX5APTD5D4/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-09T19:34:11Z","links":{"resolver":"https://pith.science/pith/WL7EVLFBMZSZMQN7YX5APTD5D4","bundle":"https://pith.science/pith/WL7EVLFBMZSZMQN7YX5APTD5D4/bundle.json","state":"https://pith.science/pith/WL7EVLFBMZSZMQN7YX5APTD5D4/state.json","well_known_bundle":"https://pith.science/.well-known/pith/WL7EVLFBMZSZMQN7YX5APTD5D4/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2020:WL7EVLFBMZSZMQN7YX5APTD5D4","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"da19a77e56762f2950411c9973d4768b2c15255795c9a630e6127063835b0b4b","cross_cats_sorted":["cs.AI","cs.RO","stat.ML"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2020-02-25T18:36:31Z","title_canon_sha256":"5c6edeba7e7fe6323d2bf5f7a3bcee6214be887c2d052c1d773e3483466d6d93"},"schema_version":"1.0","source":{"id":"2002.11089","kind":"arxiv","version":1}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2002.11089","created_at":"2026-07-05T00:43:51Z"},{"alias_kind":"arxiv_version","alias_value":"2002.11089v1","created_at":"2026-07-05T00:43:51Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2002.11089","created_at":"2026-07-05T00:43:51Z"},{"alias_kind":"pith_short_12","alias_value":"WL7EVLFBMZSZ","created_at":"2026-07-05T00:43:51Z"},{"alias_kind":"pith_short_16","alias_value":"WL7EVLFBMZSZMQN7","created_at":"2026-07-05T00:43:51Z"},{"alias_kind":"pith_short_8","alias_value":"WL7EVLFB","created_at":"2026-07-05T00:43:51Z"}],"graph_snapshots":[{"event_id":"sha256:c50e442da41987be6d0cc3f9435711b148a71ef08e909e7366e1593308cca2ca","target":"graph","created_at":"2026-07-05T00:43:51Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2002.11089/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Multi-task reinforcement learning (RL) aims to simultaneously learn policies for solving many tasks. Several prior works have found that relabeling past experience with different reward functions can improve sample efficiency. Relabeling methods typically ask: if, in hindsight, we assume that our experience was optimal for some task, for what task was it optimal? In this paper, we show that hindsight relabeling is inverse RL, an observation that suggests that we can use inverse RL in tandem for RL algorithms to efficiently solve many tasks. We use this idea to generalize goal-relabeling techni","authors_text":"Benjamin Eysenbach, Ruslan Salakhutdinov, Sergey Levine, Xinyang Geng","cross_cats":["cs.AI","cs.RO","stat.ML"],"headline":"","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2020-02-25T18:36:31Z","title":"Rewriting History with Inverse RL: Hindsight Inference for Policy Improvement"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2002.11089","kind":"arxiv","version":1},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:40da3af8b59a0ba224b8df1c0dfd27749a9b43f40dbe3ce597356e3687b79a2a","target":"record","created_at":"2026-07-05T00:43:51Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"da19a77e56762f2950411c9973d4768b2c15255795c9a630e6127063835b0b4b","cross_cats_sorted":["cs.AI","cs.RO","stat.ML"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2020-02-25T18:36:31Z","title_canon_sha256":"5c6edeba7e7fe6323d2bf5f7a3bcee6214be887c2d052c1d773e3483466d6d93"},"schema_version":"1.0","source":{"id":"2002.11089","kind":"arxiv","version":1}},"canonical_sha256":"b2fe4aaca166659641bfc5fa07cc7d1f2c0894fcf403f34fb352de6999d17dde","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"b2fe4aaca166659641bfc5fa07cc7d1f2c0894fcf403f34fb352de6999d17dde","first_computed_at":"2026-07-05T00:43:51.490990Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T00:43:51.490990Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"hA6dyVIaiWeZJmlx2O2sEtzM+BT/ty0+EHJ3rQwcZZPqf/3xzaZHTqwyNLEJB/fiuZv3t75zqahhxeOjJQDDCQ==","signature_status":"signed_v1","signed_at":"2026-07-05T00:43:51.491416Z","signed_message":"canonical_sha256_bytes"},"source_id":"2002.11089","source_kind":"arxiv","source_version":1}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:40da3af8b59a0ba224b8df1c0dfd27749a9b43f40dbe3ce597356e3687b79a2a","sha256:c50e442da41987be6d0cc3f9435711b148a71ef08e909e7366e1593308cca2ca"],"state_sha256":"4a21729e873b0ce436376cc6a102825e64392afba837cc68f20c63727096b713"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"d4NnO678wbGvld/oxej0/PhYTGCs1y0KyJmzLkSV6zugWutXFDwTcFeg03VTtlT7hCcfxdDdDbAzQnQcqtA1Dw==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-09T19:34:11.799959Z","bundle_sha256":"605ef669648d857af42c4f2883f0ccc775612e4f537d7ad8825fdb01562c99a7"}}