{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2024:KOY6XSAZOXHTFU4HFPS4FBWWLV","short_pith_number":"pith:KOY6XSAZ","canonical_record":{"source":{"id":"2403.19443","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-03-28T14:15:10Z","cross_cats_sorted":[],"title_canon_sha256":"62d115003dfcd3c5d7fdb53e1ac68e35a6cb31ffdda8c3ff78a41d95a35522a8","abstract_canon_sha256":"9ac2b734f695d3da3621e665b89f764be640092f87a6837dec79141cc9b6cf29"},"schema_version":"1.0"},"canonical_sha256":"53b1ebc81975cf32d3872be5c286d65d5c6b6c39a657d7c29d003e2d6ffe8f5d","source":{"kind":"arxiv","id":"2403.19443","version":2},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2403.19443","created_at":"2026-07-05T10:03:41Z"},{"alias_kind":"arxiv_version","alias_value":"2403.19443v2","created_at":"2026-07-05T10:03:41Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2403.19443","created_at":"2026-07-05T10:03:41Z"},{"alias_kind":"pith_short_12","alias_value":"KOY6XSAZOXHT","created_at":"2026-07-05T10:03:41Z"},{"alias_kind":"pith_short_16","alias_value":"KOY6XSAZOXHTFU4H","created_at":"2026-07-05T10:03:41Z"},{"alias_kind":"pith_short_8","alias_value":"KOY6XSAZ","created_at":"2026-07-05T10:03:41Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2024:KOY6XSAZOXHTFU4HFPS4FBWWLV","target":"record","payload":{"canonical_record":{"source":{"id":"2403.19443","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-03-28T14:15:10Z","cross_cats_sorted":[],"title_canon_sha256":"62d115003dfcd3c5d7fdb53e1ac68e35a6cb31ffdda8c3ff78a41d95a35522a8","abstract_canon_sha256":"9ac2b734f695d3da3621e665b89f764be640092f87a6837dec79141cc9b6cf29"},"schema_version":"1.0"},"canonical_sha256":"53b1ebc81975cf32d3872be5c286d65d5c6b6c39a657d7c29d003e2d6ffe8f5d","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:03:41.743126Z","signature_b64":"kjcPCKVvjxY9jOGt5PS+Csr/g7rTW5C86XkxSdpZKDcWqKx2BcjI4cXlJZptsHDeRQ0sOhCykZ45s4+QtGiOCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"53b1ebc81975cf32d3872be5c286d65d5c6b6c39a657d7c29d003e2d6ffe8f5d","last_reissued_at":"2026-07-05T10:03:41.742582Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:03:41.742582Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2403.19443","source_version":2,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T10:03:41Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"bix/CCqMZq8njVxkPrduF/kGh15aIR6TmpZgBRz9HbHg608wQCu4VwYbzQ3+GQZiRhxgEbb2IUncdsgGFhZaDw==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-07T12:12:41.545129Z"},"content_sha256":"53b23dd8ec55b4d716e80ec6ccf1977255f420c174406944007dffe7989caea9","schema_version":"1.0","event_id":"sha256:53b23dd8ec55b4d716e80ec6ccf1977255f420c174406944007dffe7989caea9"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2024:KOY6XSAZOXHTFU4HFPS4FBWWLV","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Mixed Preference Optimization: Reinforcement Learning with Data Selection and Better Reference Model","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Cam-Tu Nguyen, Qi Gou","submitted_at":"2024-03-28T14:15:10Z","abstract_excerpt":"Large Language Models (LLMs) have become increasingly popular due to their ability to process and generate natural language. However, as they are trained on massive datasets of text, LLMs can inherit harmful biases and produce outputs that are not aligned with human values. This paper studies two main approaches to LLM alignment: Reinforcement Learning with Human Feedback (RLHF) and contrastive learning-based methods like Direct Preference Optimization (DPO). By analyzing the stability and robustness of RLHF and DPO, we propose MPO (Mixed Preference Optimization), a novel method that mitigates"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2403.19443","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2403.19443/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T10:03:41Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"srMvInQNmtpJRmFw+26erRO4z4KHTe2RYSFpzIUZD9w/oPXWnLKwJf97UGsw3o77lNt1dhO3Fv+kIoelwdEDDw==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-07T12:12:41.546388Z"},"content_sha256":"e427585d6d2a887323e942963f844f57b2affe47127eb73daa6a6fc30930edd3","schema_version":"1.0","event_id":"sha256:e427585d6d2a887323e942963f844f57b2affe47127eb73daa6a6fc30930edd3"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/KOY6XSAZOXHTFU4HFPS4FBWWLV/bundle.json","state_url":"https://pith.science/pith/KOY6XSAZOXHTFU4HFPS4FBWWLV/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/KOY6XSAZOXHTFU4HFPS4FBWWLV/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-07T12:12:41Z","links":{"resolver":"https://pith.science/pith/KOY6XSAZOXHTFU4HFPS4FBWWLV","bundle":"https://pith.science/pith/KOY6XSAZOXHTFU4HFPS4FBWWLV/bundle.json","state":"https://pith.science/pith/KOY6XSAZOXHTFU4HFPS4FBWWLV/state.json","well_known_bundle":"https://pith.science/.well-known/pith/KOY6XSAZOXHTFU4HFPS4FBWWLV/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2024:KOY6XSAZOXHTFU4HFPS4FBWWLV","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"9ac2b734f695d3da3621e665b89f764be640092f87a6837dec79141cc9b6cf29","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-03-28T14:15:10Z","title_canon_sha256":"62d115003dfcd3c5d7fdb53e1ac68e35a6cb31ffdda8c3ff78a41d95a35522a8"},"schema_version":"1.0","source":{"id":"2403.19443","kind":"arxiv","version":2}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2403.19443","created_at":"2026-07-05T10:03:41Z"},{"alias_kind":"arxiv_version","alias_value":"2403.19443v2","created_at":"2026-07-05T10:03:41Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2403.19443","created_at":"2026-07-05T10:03:41Z"},{"alias_kind":"pith_short_12","alias_value":"KOY6XSAZOXHT","created_at":"2026-07-05T10:03:41Z"},{"alias_kind":"pith_short_16","alias_value":"KOY6XSAZOXHTFU4H","created_at":"2026-07-05T10:03:41Z"},{"alias_kind":"pith_short_8","alias_value":"KOY6XSAZ","created_at":"2026-07-05T10:03:41Z"}],"graph_snapshots":[{"event_id":"sha256:e427585d6d2a887323e942963f844f57b2affe47127eb73daa6a6fc30930edd3","target":"graph","created_at":"2026-07-05T10:03:41Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2403.19443/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Large Language Models (LLMs) have become increasingly popular due to their ability to process and generate natural language. However, as they are trained on massive datasets of text, LLMs can inherit harmful biases and produce outputs that are not aligned with human values. This paper studies two main approaches to LLM alignment: Reinforcement Learning with Human Feedback (RLHF) and contrastive learning-based methods like Direct Preference Optimization (DPO). By analyzing the stability and robustness of RLHF and DPO, we propose MPO (Mixed Preference Optimization), a novel method that mitigates","authors_text":"Cam-Tu Nguyen, Qi Gou","cross_cats":[],"headline":"","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-03-28T14:15:10Z","title":"Mixed Preference Optimization: Reinforcement Learning with Data Selection and Better Reference Model"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2403.19443","kind":"arxiv","version":2},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:53b23dd8ec55b4d716e80ec6ccf1977255f420c174406944007dffe7989caea9","target":"record","created_at":"2026-07-05T10:03:41Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"9ac2b734f695d3da3621e665b89f764be640092f87a6837dec79141cc9b6cf29","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-03-28T14:15:10Z","title_canon_sha256":"62d115003dfcd3c5d7fdb53e1ac68e35a6cb31ffdda8c3ff78a41d95a35522a8"},"schema_version":"1.0","source":{"id":"2403.19443","kind":"arxiv","version":2}},"canonical_sha256":"53b1ebc81975cf32d3872be5c286d65d5c6b6c39a657d7c29d003e2d6ffe8f5d","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"53b1ebc81975cf32d3872be5c286d65d5c6b6c39a657d7c29d003e2d6ffe8f5d","first_computed_at":"2026-07-05T10:03:41.742582Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T10:03:41.742582Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"kjcPCKVvjxY9jOGt5PS+Csr/g7rTW5C86XkxSdpZKDcWqKx2BcjI4cXlJZptsHDeRQ0sOhCykZ45s4+QtGiOCg==","signature_status":"signed_v1","signed_at":"2026-07-05T10:03:41.743126Z","signed_message":"canonical_sha256_bytes"},"source_id":"2403.19443","source_kind":"arxiv","source_version":2}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:53b23dd8ec55b4d716e80ec6ccf1977255f420c174406944007dffe7989caea9","sha256:e427585d6d2a887323e942963f844f57b2affe47127eb73daa6a6fc30930edd3"],"state_sha256":"828ae132892c592510143d674bf177a0ba537209e3713ed5716c0301b3e94135"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"b9a3erL+UexAiUJtLDPZM5G0P1dfS8KKr1/hlRQWVGvBrIVHUBvpNHmjFiviXz8biEkc0SZ42KhtcTbto8z3Cg==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-07T12:12:41.556164Z","bundle_sha256":"b9c47a027b3de018b1c03deca5ddd670cb09cee841653845fe5db91372950117"}}