{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2023:ADWV56HXTBCP2SV3DO44FQ7RMX","short_pith_number":"pith:ADWV56HX","canonical_record":{"source":{"id":"2305.16217","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2023-05-25T16:24:11Z","cross_cats_sorted":[],"title_canon_sha256":"6b65beb1cabf20aa67c17bcca2c6b0a2ae24ae5d9388df476d6b2c5f971acbe3","abstract_canon_sha256":"9b617ae6f6323bc263e73cc4111ae0a4ae39079a48c091e051f0220394b35269"},"schema_version":"1.0"},"canonical_sha256":"00ed5ef8f79844fd4abb1bb9c2c3f165c4247da13b80122c5dc0073134ff1be3","source":{"kind":"arxiv","id":"2305.16217","version":2},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2305.16217","created_at":"2026-07-05T06:19:04Z"},{"alias_kind":"arxiv_version","alias_value":"2305.16217v2","created_at":"2026-07-05T06:19:04Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2305.16217","created_at":"2026-07-05T06:19:04Z"},{"alias_kind":"pith_short_12","alias_value":"ADWV56HXTBCP","created_at":"2026-07-05T06:19:04Z"},{"alias_kind":"pith_short_16","alias_value":"ADWV56HXTBCP2SV3","created_at":"2026-07-05T06:19:04Z"},{"alias_kind":"pith_short_8","alias_value":"ADWV56HX","created_at":"2026-07-05T06:19:04Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2023:ADWV56HXTBCP2SV3DO44FQ7RMX","target":"record","payload":{"canonical_record":{"source":{"id":"2305.16217","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2023-05-25T16:24:11Z","cross_cats_sorted":[],"title_canon_sha256":"6b65beb1cabf20aa67c17bcca2c6b0a2ae24ae5d9388df476d6b2c5f971acbe3","abstract_canon_sha256":"9b617ae6f6323bc263e73cc4111ae0a4ae39079a48c091e051f0220394b35269"},"schema_version":"1.0"},"canonical_sha256":"00ed5ef8f79844fd4abb1bb9c2c3f165c4247da13b80122c5dc0073134ff1be3","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:19:04.401721Z","signature_b64":"gjkcVf8G/LzFS9XCzeSwIm0IUHF49NLqwqvI6BrqYJvgF5UGQv9wmBZSVkEP7mNMun2s5DgwqLmsnbZuEiH9Dg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"00ed5ef8f79844fd4abb1bb9c2c3f165c4247da13b80122c5dc0073134ff1be3","last_reissued_at":"2026-07-05T06:19:04.401238Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:19:04.401238Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2305.16217","source_version":2,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T06:19:04Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"nreF1VeUyFBXveG2BQfts48JyMIIiFiS0kuojMgBRhkk12JtVfRrFQKSu+jPrst0MIlTWoyBBkPz/7MXXm1AAQ==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-08T20:20:38.287470Z"},"content_sha256":"2c77ef7c5b096d666f8d51aa2fb18df3e597812fcb0fae1e03168ae14b874715","schema_version":"1.0","event_id":"sha256:2c77ef7c5b096d666f8d51aa2fb18df3e597812fcb0fae1e03168ae14b874715"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2023:ADWV56HXTBCP2SV3DO44FQ7RMX","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Beyond Reward: Offline Preference-guided Policy Optimization","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Diyuan Shi, Donglin Wang, Jinxin Liu, Li He, Yachen Kang","submitted_at":"2023-05-25T16:24:11Z","abstract_excerpt":"This study focuses on the topic of offline preference-based reinforcement learning (PbRL), a variant of conventional reinforcement learning that dispenses with the need for online interaction or specification of reward functions. Instead, the agent is provided with fixed offline trajectories and human preferences between pairs of trajectories to extract the dynamics and task information, respectively. Since the dynamics and task information are orthogonal, a naive approach would involve using preference-based reward learning followed by an off-the-shelf offline RL algorithm. However, this requ"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2305.16217","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2305.16217/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T06:19:04Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"BTQzOPIdl5eqK38PnhoGxiYin+kB0wIoMFwEB3RrI5gceMa4KkQkYSMG3sETR2+ijdG0Kmf02J1GPLclx2fXBQ==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-08T20:20:38.287965Z"},"content_sha256":"96515853f1e5033c3f9ac4d592418329c3146d9e9aaf268fd62fdcf8469619bd","schema_version":"1.0","event_id":"sha256:96515853f1e5033c3f9ac4d592418329c3146d9e9aaf268fd62fdcf8469619bd"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/ADWV56HXTBCP2SV3DO44FQ7RMX/bundle.json","state_url":"https://pith.science/pith/ADWV56HXTBCP2SV3DO44FQ7RMX/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/ADWV56HXTBCP2SV3DO44FQ7RMX/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-08T20:20:38Z","links":{"resolver":"https://pith.science/pith/ADWV56HXTBCP2SV3DO44FQ7RMX","bundle":"https://pith.science/pith/ADWV56HXTBCP2SV3DO44FQ7RMX/bundle.json","state":"https://pith.science/pith/ADWV56HXTBCP2SV3DO44FQ7RMX/state.json","well_known_bundle":"https://pith.science/.well-known/pith/ADWV56HXTBCP2SV3DO44FQ7RMX/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2023:ADWV56HXTBCP2SV3DO44FQ7RMX","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"9b617ae6f6323bc263e73cc4111ae0a4ae39079a48c091e051f0220394b35269","cross_cats_sorted":[],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2023-05-25T16:24:11Z","title_canon_sha256":"6b65beb1cabf20aa67c17bcca2c6b0a2ae24ae5d9388df476d6b2c5f971acbe3"},"schema_version":"1.0","source":{"id":"2305.16217","kind":"arxiv","version":2}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2305.16217","created_at":"2026-07-05T06:19:04Z"},{"alias_kind":"arxiv_version","alias_value":"2305.16217v2","created_at":"2026-07-05T06:19:04Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2305.16217","created_at":"2026-07-05T06:19:04Z"},{"alias_kind":"pith_short_12","alias_value":"ADWV56HXTBCP","created_at":"2026-07-05T06:19:04Z"},{"alias_kind":"pith_short_16","alias_value":"ADWV56HXTBCP2SV3","created_at":"2026-07-05T06:19:04Z"},{"alias_kind":"pith_short_8","alias_value":"ADWV56HX","created_at":"2026-07-05T06:19:04Z"}],"graph_snapshots":[{"event_id":"sha256:96515853f1e5033c3f9ac4d592418329c3146d9e9aaf268fd62fdcf8469619bd","target":"graph","created_at":"2026-07-05T06:19:04Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2305.16217/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"This study focuses on the topic of offline preference-based reinforcement learning (PbRL), a variant of conventional reinforcement learning that dispenses with the need for online interaction or specification of reward functions. Instead, the agent is provided with fixed offline trajectories and human preferences between pairs of trajectories to extract the dynamics and task information, respectively. Since the dynamics and task information are orthogonal, a naive approach would involve using preference-based reward learning followed by an off-the-shelf offline RL algorithm. However, this requ","authors_text":"Diyuan Shi, Donglin Wang, Jinxin Liu, Li He, Yachen Kang","cross_cats":[],"headline":"","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2023-05-25T16:24:11Z","title":"Beyond Reward: Offline Preference-guided Policy Optimization"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2305.16217","kind":"arxiv","version":2},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:2c77ef7c5b096d666f8d51aa2fb18df3e597812fcb0fae1e03168ae14b874715","target":"record","created_at":"2026-07-05T06:19:04Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"9b617ae6f6323bc263e73cc4111ae0a4ae39079a48c091e051f0220394b35269","cross_cats_sorted":[],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2023-05-25T16:24:11Z","title_canon_sha256":"6b65beb1cabf20aa67c17bcca2c6b0a2ae24ae5d9388df476d6b2c5f971acbe3"},"schema_version":"1.0","source":{"id":"2305.16217","kind":"arxiv","version":2}},"canonical_sha256":"00ed5ef8f79844fd4abb1bb9c2c3f165c4247da13b80122c5dc0073134ff1be3","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"00ed5ef8f79844fd4abb1bb9c2c3f165c4247da13b80122c5dc0073134ff1be3","first_computed_at":"2026-07-05T06:19:04.401238Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T06:19:04.401238Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"gjkcVf8G/LzFS9XCzeSwIm0IUHF49NLqwqvI6BrqYJvgF5UGQv9wmBZSVkEP7mNMun2s5DgwqLmsnbZuEiH9Dg==","signature_status":"signed_v1","signed_at":"2026-07-05T06:19:04.401721Z","signed_message":"canonical_sha256_bytes"},"source_id":"2305.16217","source_kind":"arxiv","source_version":2}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:2c77ef7c5b096d666f8d51aa2fb18df3e597812fcb0fae1e03168ae14b874715","sha256:96515853f1e5033c3f9ac4d592418329c3146d9e9aaf268fd62fdcf8469619bd"],"state_sha256":"9676f65c34221c7097aa1c468c95304b4e9f8ab7097144bf539350079d0dccf1"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"7KWipTjJae2t1o+SNYjy7ZZlXEKweCPdV0yEvdrWQsDd7h6ZGCXr/XtAt3cxMsVee9Gxr5VSgi3lgveTf7C4Cw==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-08T20:20:38.291712Z","bundle_sha256":"aed12f72ba28854ead8cba89fd4695101afaa3f60516b092d4201e34bddfed64"}}