{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2023:7NVZTC7T6QSXJYUIH4RQH2PL3I","short_pith_number":"pith:7NVZTC7T","canonical_record":{"source":{"id":"2306.13085","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2023-06-22T17:58:02Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"74ce84b4e653c8c59bb6bcc5f5391c17cb0857d23ab1d8eaa38922874f44c7cf","abstract_canon_sha256":"002e72918315b985e2928624a80e4a3a1cac3b649ab6fde232da0ca59497bbf0"},"schema_version":"1.0"},"canonical_sha256":"fb6b998bf3f42574e2883f2303e9ebda0ab1c044358ee7afd925071f0c349b62","source":{"kind":"arxiv","id":"2306.13085","version":1},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2306.13085","created_at":"2026-07-05T06:23:51Z"},{"alias_kind":"arxiv_version","alias_value":"2306.13085v1","created_at":"2026-07-05T06:23:51Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2306.13085","created_at":"2026-07-05T06:23:51Z"},{"alias_kind":"pith_short_12","alias_value":"7NVZTC7T6QSX","created_at":"2026-07-05T06:23:51Z"},{"alias_kind":"pith_short_16","alias_value":"7NVZTC7T6QSXJYUI","created_at":"2026-07-05T06:23:51Z"},{"alias_kind":"pith_short_8","alias_value":"7NVZTC7T","created_at":"2026-07-05T06:23:51Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2023:7NVZTC7T6QSXJYUIH4RQH2PL3I","target":"record","payload":{"canonical_record":{"source":{"id":"2306.13085","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2023-06-22T17:58:02Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"74ce84b4e653c8c59bb6bcc5f5391c17cb0857d23ab1d8eaa38922874f44c7cf","abstract_canon_sha256":"002e72918315b985e2928624a80e4a3a1cac3b649ab6fde232da0ca59497bbf0"},"schema_version":"1.0"},"canonical_sha256":"fb6b998bf3f42574e2883f2303e9ebda0ab1c044358ee7afd925071f0c349b62","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:23:51.909464Z","signature_b64":"suB5BwMgNf4PqQnafgBlMiSCcM0yeOTQrtyybwprlPkRdDTwEAqpGU3rijwbet8AlPSCaYlcyqRzjM45fAIdBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"fb6b998bf3f42574e2883f2303e9ebda0ab1c044358ee7afd925071f0c349b62","last_reissued_at":"2026-07-05T06:23:51.909043Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:23:51.909043Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2306.13085","source_version":1,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T06:23:51Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"TgRJ6wWxdysKA92jpHbZ7O8i8Kw04PdAqUr95S6BO39vpsTyiyOSpOSEzbiinymorW5oSPXXrV9CMU4C0y7GBQ==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-08T17:43:04.217656Z"},"content_sha256":"aac6cd26ea8f0b1c6c23e8eff03b6819dff38116b069196647c8372c74bba4ab","schema_version":"1.0","event_id":"sha256:aac6cd26ea8f0b1c6c23e8eff03b6819dff38116b069196647c8372c74bba4ab"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2023:7NVZTC7T6QSXJYUIH4RQH2PL3I","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Harnessing Mixed Offline Reinforcement Learning Datasets via Trajectory Weighting","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Pulkit Agrawal, R\\'emi Tachet des Combes, Romain Laroche, Zhang-Wei Hong","submitted_at":"2023-06-22T17:58:02Z","abstract_excerpt":"Most offline reinforcement learning (RL) algorithms return a target policy maximizing a trade-off between (1) the expected performance gain over the behavior policy that collected the dataset, and (2) the risk stemming from the out-of-distribution-ness of the induced state-action occupancy. It follows that the performance of the target policy is strongly related to the performance of the behavior policy and, thus, the trajectory return distribution of the dataset. We show that in mixed datasets consisting of mostly low-return trajectories and minor high-return trajectories, state-of-the-art of"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2306.13085","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2306.13085/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T06:23:51Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"HCECcOqbryKfLO70mCciqIapWm9YNjrLVRNyGXiLqtiz1acYTJJg02qfvgkaZ8x2RwQTumjBxiI6aywNGqEqAA==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-08T17:43:04.218167Z"},"content_sha256":"54770c436c34a7880a96097e987b8544f2edcfd0d0dc2f4804bf5c718f6b8b87","schema_version":"1.0","event_id":"sha256:54770c436c34a7880a96097e987b8544f2edcfd0d0dc2f4804bf5c718f6b8b87"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/7NVZTC7T6QSXJYUIH4RQH2PL3I/bundle.json","state_url":"https://pith.science/pith/7NVZTC7T6QSXJYUIH4RQH2PL3I/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/7NVZTC7T6QSXJYUIH4RQH2PL3I/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-08T17:43:04Z","links":{"resolver":"https://pith.science/pith/7NVZTC7T6QSXJYUIH4RQH2PL3I","bundle":"https://pith.science/pith/7NVZTC7T6QSXJYUIH4RQH2PL3I/bundle.json","state":"https://pith.science/pith/7NVZTC7T6QSXJYUIH4RQH2PL3I/state.json","well_known_bundle":"https://pith.science/.well-known/pith/7NVZTC7T6QSXJYUIH4RQH2PL3I/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2023:7NVZTC7T6QSXJYUIH4RQH2PL3I","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"002e72918315b985e2928624a80e4a3a1cac3b649ab6fde232da0ca59497bbf0","cross_cats_sorted":["cs.AI"],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2023-06-22T17:58:02Z","title_canon_sha256":"74ce84b4e653c8c59bb6bcc5f5391c17cb0857d23ab1d8eaa38922874f44c7cf"},"schema_version":"1.0","source":{"id":"2306.13085","kind":"arxiv","version":1}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2306.13085","created_at":"2026-07-05T06:23:51Z"},{"alias_kind":"arxiv_version","alias_value":"2306.13085v1","created_at":"2026-07-05T06:23:51Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2306.13085","created_at":"2026-07-05T06:23:51Z"},{"alias_kind":"pith_short_12","alias_value":"7NVZTC7T6QSX","created_at":"2026-07-05T06:23:51Z"},{"alias_kind":"pith_short_16","alias_value":"7NVZTC7T6QSXJYUI","created_at":"2026-07-05T06:23:51Z"},{"alias_kind":"pith_short_8","alias_value":"7NVZTC7T","created_at":"2026-07-05T06:23:51Z"}],"graph_snapshots":[{"event_id":"sha256:54770c436c34a7880a96097e987b8544f2edcfd0d0dc2f4804bf5c718f6b8b87","target":"graph","created_at":"2026-07-05T06:23:51Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2306.13085/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Most offline reinforcement learning (RL) algorithms return a target policy maximizing a trade-off between (1) the expected performance gain over the behavior policy that collected the dataset, and (2) the risk stemming from the out-of-distribution-ness of the induced state-action occupancy. It follows that the performance of the target policy is strongly related to the performance of the behavior policy and, thus, the trajectory return distribution of the dataset. We show that in mixed datasets consisting of mostly low-return trajectories and minor high-return trajectories, state-of-the-art of","authors_text":"Pulkit Agrawal, R\\'emi Tachet des Combes, Romain Laroche, Zhang-Wei Hong","cross_cats":["cs.AI"],"headline":"","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2023-06-22T17:58:02Z","title":"Harnessing Mixed Offline Reinforcement Learning Datasets via Trajectory Weighting"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2306.13085","kind":"arxiv","version":1},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:aac6cd26ea8f0b1c6c23e8eff03b6819dff38116b069196647c8372c74bba4ab","target":"record","created_at":"2026-07-05T06:23:51Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"002e72918315b985e2928624a80e4a3a1cac3b649ab6fde232da0ca59497bbf0","cross_cats_sorted":["cs.AI"],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2023-06-22T17:58:02Z","title_canon_sha256":"74ce84b4e653c8c59bb6bcc5f5391c17cb0857d23ab1d8eaa38922874f44c7cf"},"schema_version":"1.0","source":{"id":"2306.13085","kind":"arxiv","version":1}},"canonical_sha256":"fb6b998bf3f42574e2883f2303e9ebda0ab1c044358ee7afd925071f0c349b62","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"fb6b998bf3f42574e2883f2303e9ebda0ab1c044358ee7afd925071f0c349b62","first_computed_at":"2026-07-05T06:23:51.909043Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T06:23:51.909043Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"suB5BwMgNf4PqQnafgBlMiSCcM0yeOTQrtyybwprlPkRdDTwEAqpGU3rijwbet8AlPSCaYlcyqRzjM45fAIdBg==","signature_status":"signed_v1","signed_at":"2026-07-05T06:23:51.909464Z","signed_message":"canonical_sha256_bytes"},"source_id":"2306.13085","source_kind":"arxiv","source_version":1}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:aac6cd26ea8f0b1c6c23e8eff03b6819dff38116b069196647c8372c74bba4ab","sha256:54770c436c34a7880a96097e987b8544f2edcfd0d0dc2f4804bf5c718f6b8b87"],"state_sha256":"1df8d51348904d4ac3a7386a616588f66192f4423cfc53464e103280e02543f2"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"qd2I9L443di/Ri7kBG4fNDxs0PorCZ5mULriIAMCbHn8ZagGKWwrxegaICuK6InBDpdmqG/LVb7pRwPll0n4DQ==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-08T17:43:04.228681Z","bundle_sha256":"bd8d8e8164315ccaccd3c7d49ff7c8a26bfe4bc0c982a6bf59a50b777177822d"}}