{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:ADWV56HXTBCP2SV3DO44FQ7RMX","short_pith_number":"pith:ADWV56HX","schema_version":"1.0","canonical_sha256":"00ed5ef8f79844fd4abb1bb9c2c3f165c4247da13b80122c5dc0073134ff1be3","source":{"kind":"arxiv","id":"2305.16217","version":2},"attestation_state":"computed","paper":{"title":"Beyond Reward: Offline Preference-guided Policy Optimization","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Diyuan Shi, Donglin Wang, Jinxin Liu, Li He, Yachen Kang","submitted_at":"2023-05-25T16:24:11Z","abstract_excerpt":"This study focuses on the topic of offline preference-based reinforcement learning (PbRL), a variant of conventional reinforcement learning that dispenses with the need for online interaction or specification of reward functions. Instead, the agent is provided with fixed offline trajectories and human preferences between pairs of trajectories to extract the dynamics and task information, respectively. Since the dynamics and task information are orthogonal, a naive approach would involve using preference-based reward learning followed by an off-the-shelf offline RL algorithm. However, this requ"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2305.16217","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2023-05-25T16:24:11Z","cross_cats_sorted":[],"title_canon_sha256":"6b65beb1cabf20aa67c17bcca2c6b0a2ae24ae5d9388df476d6b2c5f971acbe3","abstract_canon_sha256":"9b617ae6f6323bc263e73cc4111ae0a4ae39079a48c091e051f0220394b35269"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:19:04.401721Z","signature_b64":"gjkcVf8G/LzFS9XCzeSwIm0IUHF49NLqwqvI6BrqYJvgF5UGQv9wmBZSVkEP7mNMun2s5DgwqLmsnbZuEiH9Dg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"00ed5ef8f79844fd4abb1bb9c2c3f165c4247da13b80122c5dc0073134ff1be3","last_reissued_at":"2026-07-05T06:19:04.401238Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:19:04.401238Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Beyond Reward: Offline Preference-guided Policy Optimization","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Diyuan Shi, Donglin Wang, Jinxin Liu, Li He, Yachen Kang","submitted_at":"2023-05-25T16:24:11Z","abstract_excerpt":"This study focuses on the topic of offline preference-based reinforcement learning (PbRL), a variant of conventional reinforcement learning that dispenses with the need for online interaction or specification of reward functions. Instead, the agent is provided with fixed offline trajectories and human preferences between pairs of trajectories to extract the dynamics and task information, respectively. Since the dynamics and task information are orthogonal, a naive approach would involve using preference-based reward learning followed by an off-the-shelf offline RL algorithm. However, this requ"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2305.16217","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2305.16217/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2305.16217","created_at":"2026-07-05T06:19:04.401295+00:00"},{"alias_kind":"arxiv_version","alias_value":"2305.16217v2","created_at":"2026-07-05T06:19:04.401295+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2305.16217","created_at":"2026-07-05T06:19:04.401295+00:00"},{"alias_kind":"pith_short_12","alias_value":"ADWV56HXTBCP","created_at":"2026-07-05T06:19:04.401295+00:00"},{"alias_kind":"pith_short_16","alias_value":"ADWV56HXTBCP2SV3","created_at":"2026-07-05T06:19:04.401295+00:00"},{"alias_kind":"pith_short_8","alias_value":"ADWV56HX","created_at":"2026-07-05T06:19:04.401295+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2604.02349","citing_title":"OPRIDE: Offline Preference-based Reinforcement Learning via In-Dataset Exploration","ref_index":23,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ADWV56HXTBCP2SV3DO44FQ7RMX","json":"https://pith.science/pith/ADWV56HXTBCP2SV3DO44FQ7RMX.json","graph_json":"https://pith.science/api/pith-number/ADWV56HXTBCP2SV3DO44FQ7RMX/graph.json","events_json":"https://pith.science/api/pith-number/ADWV56HXTBCP2SV3DO44FQ7RMX/events.json","paper":"https://pith.science/paper/ADWV56HX"},"agent_actions":{"view_html":"https://pith.science/pith/ADWV56HXTBCP2SV3DO44FQ7RMX","download_json":"https://pith.science/pith/ADWV56HXTBCP2SV3DO44FQ7RMX.json","view_paper":"https://pith.science/paper/ADWV56HX","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2305.16217&json=true","fetch_graph":"https://pith.science/api/pith-number/ADWV56HXTBCP2SV3DO44FQ7RMX/graph.json","fetch_events":"https://pith.science/api/pith-number/ADWV56HXTBCP2SV3DO44FQ7RMX/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ADWV56HXTBCP2SV3DO44FQ7RMX/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ADWV56HXTBCP2SV3DO44FQ7RMX/action/storage_attestation","attest_author":"https://pith.science/pith/ADWV56HXTBCP2SV3DO44FQ7RMX/action/author_attestation","sign_citation":"https://pith.science/pith/ADWV56HXTBCP2SV3DO44FQ7RMX/action/citation_signature","submit_replication":"https://pith.science/pith/ADWV56HXTBCP2SV3DO44FQ7RMX/action/replication_record"}},"created_at":"2026-07-05T06:19:04.401295+00:00","updated_at":"2026-07-05T06:19:04.401295+00:00"}