{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2021:GWBPM7PPY45YLTFIY63WULQPQM","short_pith_number":"pith:GWBPM7PP","canonical_record":{"source":{"id":"2106.09119","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2021-06-16T20:48:49Z","cross_cats_sorted":[],"title_canon_sha256":"0c703f23fb37c1a450745567320c0570417911b3b700dd5f24be0811fd3470ed","abstract_canon_sha256":"3ec264dee59c4b06703034130f8a6fb04fd89f301986856c394f2b3a3e3250a4"},"schema_version":"1.0"},"canonical_sha256":"3582f67defc73b85cca8c7b76a2e0f83113c3fd9bb05b252a27f832db8a6785b","source":{"kind":"arxiv","id":"2106.09119","version":2},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2106.09119","created_at":"2026-07-05T02:50:30Z"},{"alias_kind":"arxiv_version","alias_value":"2106.09119v2","created_at":"2026-07-05T02:50:30Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2106.09119","created_at":"2026-07-05T02:50:30Z"},{"alias_kind":"pith_short_12","alias_value":"GWBPM7PPY45Y","created_at":"2026-07-05T02:50:30Z"},{"alias_kind":"pith_short_16","alias_value":"GWBPM7PPY45YLTFI","created_at":"2026-07-05T02:50:30Z"},{"alias_kind":"pith_short_8","alias_value":"GWBPM7PP","created_at":"2026-07-05T02:50:30Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2021:GWBPM7PPY45YLTFIY63WULQPQM","target":"record","payload":{"canonical_record":{"source":{"id":"2106.09119","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2021-06-16T20:48:49Z","cross_cats_sorted":[],"title_canon_sha256":"0c703f23fb37c1a450745567320c0570417911b3b700dd5f24be0811fd3470ed","abstract_canon_sha256":"3ec264dee59c4b06703034130f8a6fb04fd89f301986856c394f2b3a3e3250a4"},"schema_version":"1.0"},"canonical_sha256":"3582f67defc73b85cca8c7b76a2e0f83113c3fd9bb05b252a27f832db8a6785b","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T02:50:30.896370Z","signature_b64":"piDAAs/pBz1Zv1v27DqBeYyom3qJ+EFBCzrHrJ3dsDtkINUtm1iFU+ZpsZLAyn515/5mGcH5RksxEnxE27aAAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"3582f67defc73b85cca8c7b76a2e0f83113c3fd9bb05b252a27f832db8a6785b","last_reissued_at":"2026-07-05T02:50:30.896018Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T02:50:30.896018Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2106.09119","source_version":2,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T02:50:30Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"V8Fl0cTkghIOuCBEogC7I8n83iExASLPEzpxfwgGIVjJHNBl+arD5qr3DMOwbATIDFgq5+6Bg1FAq8kJG+iDCw==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-06T09:54:44.139106Z"},"content_sha256":"1067b1827ea5c902c72b9be227c65b18efb60c85d05e2ee656b9796a3e6a7abb","schema_version":"1.0","event_id":"sha256:1067b1827ea5c902c72b9be227c65b18efb60c85d05e2ee656b9796a3e6a7abb"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2021:GWBPM7PPY45YLTFIY63WULQPQM","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Behavioral Priors and Dynamics Models: Improving Performance and Domain Transfer in Offline RL","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Aravind Rajeswaran, Catherine Cang, Michael Laskin, Pieter Abbeel","submitted_at":"2021-06-16T20:48:49Z","abstract_excerpt":"Offline Reinforcement Learning (RL) aims to extract near-optimal policies from imperfect offline data without additional environment interactions. Extracting policies from diverse offline datasets has the potential to expand the range of applicability of RL by making the training process safer, faster, and more streamlined. We investigate how to improve the performance of offline RL algorithms, its robustness to the quality of offline data, as well as its generalization capabilities. To this end, we introduce Offline Model-based RL with Adaptive Behavioral Priors (MABE). Our algorithm is based"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2106.09119","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2106.09119/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T02:50:30Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"yt8Eba3n5MhN0B/tvckt5wtAJ569UXexvck88yfpyJMKnvpGkaknO3Wqrsu2yeS0JD2TNYl4GqY8nzSAukPBDA==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-06T09:54:44.139740Z"},"content_sha256":"e273a789a956d67edfe38b825c1bb7a6fa4fdcc3fea1a19486bf70f1c1a0e0d5","schema_version":"1.0","event_id":"sha256:e273a789a956d67edfe38b825c1bb7a6fa4fdcc3fea1a19486bf70f1c1a0e0d5"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/GWBPM7PPY45YLTFIY63WULQPQM/bundle.json","state_url":"https://pith.science/pith/GWBPM7PPY45YLTFIY63WULQPQM/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/GWBPM7PPY45YLTFIY63WULQPQM/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-06T09:54:44Z","links":{"resolver":"https://pith.science/pith/GWBPM7PPY45YLTFIY63WULQPQM","bundle":"https://pith.science/pith/GWBPM7PPY45YLTFIY63WULQPQM/bundle.json","state":"https://pith.science/pith/GWBPM7PPY45YLTFIY63WULQPQM/state.json","well_known_bundle":"https://pith.science/.well-known/pith/GWBPM7PPY45YLTFIY63WULQPQM/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2021:GWBPM7PPY45YLTFIY63WULQPQM","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"3ec264dee59c4b06703034130f8a6fb04fd89f301986856c394f2b3a3e3250a4","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2021-06-16T20:48:49Z","title_canon_sha256":"0c703f23fb37c1a450745567320c0570417911b3b700dd5f24be0811fd3470ed"},"schema_version":"1.0","source":{"id":"2106.09119","kind":"arxiv","version":2}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2106.09119","created_at":"2026-07-05T02:50:30Z"},{"alias_kind":"arxiv_version","alias_value":"2106.09119v2","created_at":"2026-07-05T02:50:30Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2106.09119","created_at":"2026-07-05T02:50:30Z"},{"alias_kind":"pith_short_12","alias_value":"GWBPM7PPY45Y","created_at":"2026-07-05T02:50:30Z"},{"alias_kind":"pith_short_16","alias_value":"GWBPM7PPY45YLTFI","created_at":"2026-07-05T02:50:30Z"},{"alias_kind":"pith_short_8","alias_value":"GWBPM7PP","created_at":"2026-07-05T02:50:30Z"}],"graph_snapshots":[{"event_id":"sha256:e273a789a956d67edfe38b825c1bb7a6fa4fdcc3fea1a19486bf70f1c1a0e0d5","target":"graph","created_at":"2026-07-05T02:50:30Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2106.09119/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Offline Reinforcement Learning (RL) aims to extract near-optimal policies from imperfect offline data without additional environment interactions. Extracting policies from diverse offline datasets has the potential to expand the range of applicability of RL by making the training process safer, faster, and more streamlined. We investigate how to improve the performance of offline RL algorithms, its robustness to the quality of offline data, as well as its generalization capabilities. To this end, we introduce Offline Model-based RL with Adaptive Behavioral Priors (MABE). Our algorithm is based","authors_text":"Aravind Rajeswaran, Catherine Cang, Michael Laskin, Pieter Abbeel","cross_cats":[],"headline":"","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2021-06-16T20:48:49Z","title":"Behavioral Priors and Dynamics Models: Improving Performance and Domain Transfer in Offline RL"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2106.09119","kind":"arxiv","version":2},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:1067b1827ea5c902c72b9be227c65b18efb60c85d05e2ee656b9796a3e6a7abb","target":"record","created_at":"2026-07-05T02:50:30Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"3ec264dee59c4b06703034130f8a6fb04fd89f301986856c394f2b3a3e3250a4","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2021-06-16T20:48:49Z","title_canon_sha256":"0c703f23fb37c1a450745567320c0570417911b3b700dd5f24be0811fd3470ed"},"schema_version":"1.0","source":{"id":"2106.09119","kind":"arxiv","version":2}},"canonical_sha256":"3582f67defc73b85cca8c7b76a2e0f83113c3fd9bb05b252a27f832db8a6785b","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"3582f67defc73b85cca8c7b76a2e0f83113c3fd9bb05b252a27f832db8a6785b","first_computed_at":"2026-07-05T02:50:30.896018Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T02:50:30.896018Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"piDAAs/pBz1Zv1v27DqBeYyom3qJ+EFBCzrHrJ3dsDtkINUtm1iFU+ZpsZLAyn515/5mGcH5RksxEnxE27aAAQ==","signature_status":"signed_v1","signed_at":"2026-07-05T02:50:30.896370Z","signed_message":"canonical_sha256_bytes"},"source_id":"2106.09119","source_kind":"arxiv","source_version":2}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:1067b1827ea5c902c72b9be227c65b18efb60c85d05e2ee656b9796a3e6a7abb","sha256:e273a789a956d67edfe38b825c1bb7a6fa4fdcc3fea1a19486bf70f1c1a0e0d5"],"state_sha256":"0440e126415cccfadb0975d663a2455162c4c9e857ef5412d59ee85c2a9a9b08"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"Mk4IZfALnEvitDS/0Om6kqfGRXgM1fDTAOrXB0o/Kdp9DCNA2IOSIsdesR+8T+BHeQ9PQZoGa5f5Q8JvmhSUDg==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-06T09:54:44.143893Z","bundle_sha256":"1c398bb491fef09d0747b374cced1fff0f13b7dd82fc07c034996655c6d149c6"}}