{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2024:X6OFXYARJV6LYHA75KLGIZ74IK","short_pith_number":"pith:X6OFXYAR","canonical_record":{"source":{"id":"2403.01857","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-03-04T09:13:14Z","cross_cats_sorted":[],"title_canon_sha256":"8d83a63a76bda4d4fe5006ac731b85ec82a087eb795b4e6474ff9f291884cab8","abstract_canon_sha256":"23f53e274657691fb7cc6bd48ccda4eecb9c848877b1d92e74b817c96016b558"},"schema_version":"1.0"},"canonical_sha256":"bf9c5be0114d7cbc1c1fea966467fc4299e07686efb77936b7660d9764b6a437","source":{"kind":"arxiv","id":"2403.01857","version":2},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2403.01857","created_at":"2026-07-05T08:27:46Z"},{"alias_kind":"arxiv_version","alias_value":"2403.01857v2","created_at":"2026-07-05T08:27:46Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2403.01857","created_at":"2026-07-05T08:27:46Z"},{"alias_kind":"pith_short_12","alias_value":"X6OFXYARJV6L","created_at":"2026-07-05T08:27:46Z"},{"alias_kind":"pith_short_16","alias_value":"X6OFXYARJV6LYHA7","created_at":"2026-07-05T08:27:46Z"},{"alias_kind":"pith_short_8","alias_value":"X6OFXYAR","created_at":"2026-07-05T08:27:46Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2024:X6OFXYARJV6LYHA75KLGIZ74IK","target":"record","payload":{"canonical_record":{"source":{"id":"2403.01857","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-03-04T09:13:14Z","cross_cats_sorted":[],"title_canon_sha256":"8d83a63a76bda4d4fe5006ac731b85ec82a087eb795b4e6474ff9f291884cab8","abstract_canon_sha256":"23f53e274657691fb7cc6bd48ccda4eecb9c848877b1d92e74b817c96016b558"},"schema_version":"1.0"},"canonical_sha256":"bf9c5be0114d7cbc1c1fea966467fc4299e07686efb77936b7660d9764b6a437","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:27:46.943409Z","signature_b64":"MySCy95YusiXAqFP3QaQ9Q9OgdD0F8Q77BqmZVZacMJH6vgB26T5sT/ddF3J/nSKedpKfKDMFgEwz7PvFcAFDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"bf9c5be0114d7cbc1c1fea966467fc4299e07686efb77936b7660d9764b6a437","last_reissued_at":"2026-07-05T08:27:46.942895Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:27:46.942895Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2403.01857","source_version":2,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T08:27:46Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"Udfnp6pfcS1pczoNacy/LH2E6zsMJs71jd5c4fPz/kgfBz9cC4nygCOnCSSHvNN9YNuvLNAWj1SWSbBiMofrCQ==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-19T17:59:26.058655Z"},"content_sha256":"6df96330b67cb9ef1b60a4fe5324c85a213a532c544fae19b22e8a5aee66e1a4","schema_version":"1.0","event_id":"sha256:6df96330b67cb9ef1b60a4fe5324c85a213a532c544fae19b22e8a5aee66e1a4"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2024:X6OFXYARJV6LYHA75KLGIZ74IK","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Reward Model Learning vs. Direct Policy Optimization: A Comparative Analysis of Learning from Human Preferences","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Adish Singla, Andi Nika, Debmalya Mandal, Georgios Tzannetos, Goran Radanovi\\'c, Parameswaran Kamalaruban","submitted_at":"2024-03-04T09:13:14Z","abstract_excerpt":"In this paper, we take a step towards a deeper understanding of learning from human preferences by systematically comparing the paradigm of reinforcement learning from human feedback (RLHF) with the recently proposed paradigm of direct preference optimization (DPO). We focus our attention on the class of loglinear policy parametrization and linear reward functions. In order to compare the two paradigms, we first derive minimax statistical bounds on the suboptimality gap induced by both RLHF and DPO, assuming access to an oracle that exactly solves the optimization problems. We provide a detail"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2403.01857","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2403.01857/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T08:27:46Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"aobDD97OkdIRFsAd4KFVtFOtB7vHqhS2G/kzrf8RR9B3NFv+xu0RNzM+TSGqizJhvvS0kJCPFFOXq1qqGHvYDg==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-19T17:59:26.059026Z"},"content_sha256":"90e5efc60c09a97b7dea68cc40dab49611e975217bc62c3d61032c7bc4c6be77","schema_version":"1.0","event_id":"sha256:90e5efc60c09a97b7dea68cc40dab49611e975217bc62c3d61032c7bc4c6be77"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/X6OFXYARJV6LYHA75KLGIZ74IK/bundle.json","state_url":"https://pith.science/pith/X6OFXYARJV6LYHA75KLGIZ74IK/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/X6OFXYARJV6LYHA75KLGIZ74IK/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-19T17:59:26Z","links":{"resolver":"https://pith.science/pith/X6OFXYARJV6LYHA75KLGIZ74IK","bundle":"https://pith.science/pith/X6OFXYARJV6LYHA75KLGIZ74IK/bundle.json","state":"https://pith.science/pith/X6OFXYARJV6LYHA75KLGIZ74IK/state.json","well_known_bundle":"https://pith.science/.well-known/pith/X6OFXYARJV6LYHA75KLGIZ74IK/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2024:X6OFXYARJV6LYHA75KLGIZ74IK","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"23f53e274657691fb7cc6bd48ccda4eecb9c848877b1d92e74b817c96016b558","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-03-04T09:13:14Z","title_canon_sha256":"8d83a63a76bda4d4fe5006ac731b85ec82a087eb795b4e6474ff9f291884cab8"},"schema_version":"1.0","source":{"id":"2403.01857","kind":"arxiv","version":2}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2403.01857","created_at":"2026-07-05T08:27:46Z"},{"alias_kind":"arxiv_version","alias_value":"2403.01857v2","created_at":"2026-07-05T08:27:46Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2403.01857","created_at":"2026-07-05T08:27:46Z"},{"alias_kind":"pith_short_12","alias_value":"X6OFXYARJV6L","created_at":"2026-07-05T08:27:46Z"},{"alias_kind":"pith_short_16","alias_value":"X6OFXYARJV6LYHA7","created_at":"2026-07-05T08:27:46Z"},{"alias_kind":"pith_short_8","alias_value":"X6OFXYAR","created_at":"2026-07-05T08:27:46Z"}],"graph_snapshots":[{"event_id":"sha256:90e5efc60c09a97b7dea68cc40dab49611e975217bc62c3d61032c7bc4c6be77","target":"graph","created_at":"2026-07-05T08:27:46Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2403.01857/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"In this paper, we take a step towards a deeper understanding of learning from human preferences by systematically comparing the paradigm of reinforcement learning from human feedback (RLHF) with the recently proposed paradigm of direct preference optimization (DPO). We focus our attention on the class of loglinear policy parametrization and linear reward functions. In order to compare the two paradigms, we first derive minimax statistical bounds on the suboptimality gap induced by both RLHF and DPO, assuming access to an oracle that exactly solves the optimization problems. We provide a detail","authors_text":"Adish Singla, Andi Nika, Debmalya Mandal, Georgios Tzannetos, Goran Radanovi\\'c, Parameswaran Kamalaruban","cross_cats":[],"headline":"","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-03-04T09:13:14Z","title":"Reward Model Learning vs. Direct Policy Optimization: A Comparative Analysis of Learning from Human Preferences"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2403.01857","kind":"arxiv","version":2},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:6df96330b67cb9ef1b60a4fe5324c85a213a532c544fae19b22e8a5aee66e1a4","target":"record","created_at":"2026-07-05T08:27:46Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"23f53e274657691fb7cc6bd48ccda4eecb9c848877b1d92e74b817c96016b558","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-03-04T09:13:14Z","title_canon_sha256":"8d83a63a76bda4d4fe5006ac731b85ec82a087eb795b4e6474ff9f291884cab8"},"schema_version":"1.0","source":{"id":"2403.01857","kind":"arxiv","version":2}},"canonical_sha256":"bf9c5be0114d7cbc1c1fea966467fc4299e07686efb77936b7660d9764b6a437","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"bf9c5be0114d7cbc1c1fea966467fc4299e07686efb77936b7660d9764b6a437","first_computed_at":"2026-07-05T08:27:46.942895Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T08:27:46.942895Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"MySCy95YusiXAqFP3QaQ9Q9OgdD0F8Q77BqmZVZacMJH6vgB26T5sT/ddF3J/nSKedpKfKDMFgEwz7PvFcAFDQ==","signature_status":"signed_v1","signed_at":"2026-07-05T08:27:46.943409Z","signed_message":"canonical_sha256_bytes"},"source_id":"2403.01857","source_kind":"arxiv","source_version":2}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:6df96330b67cb9ef1b60a4fe5324c85a213a532c544fae19b22e8a5aee66e1a4","sha256:90e5efc60c09a97b7dea68cc40dab49611e975217bc62c3d61032c7bc4c6be77"],"state_sha256":"61d0efde770bb68a3b9b46150219ff2e0f02fd190a0e14dfef806d46b2effef7"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"v5CzRE2XxwU5I3GH4imtyGtX8c3YjTKgFA1MHjiCTZ2O+lBOLS0uJ00yV9he9pG9TOwuHSI/4Z2lkmgUEsmzCg==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-19T17:59:26.061542Z","bundle_sha256":"9944bfd53352c0ccc02f3b130b38cd60432d137bfa9c28c44b90c4f179b6fa79"}}