{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2025:7VGEXAL5VHE5PS7TIGZCX4UFQP","short_pith_number":"pith:7VGEXAL5","canonical_record":{"source":{"id":"2502.20847","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-02-28T08:47:03Z","cross_cats_sorted":[],"title_canon_sha256":"f7c0f99c752d9249bb2941d4b6004819fc7151312d96b4f3d41b96b3c5b897e3","abstract_canon_sha256":"6edb87d0bd995b2e3c5539a87105feb77e073fa69e7f31a2cded466df307069e"},"schema_version":"1.0"},"canonical_sha256":"fd4c4b817da9c9d7cbf341b22bf28583d8cff08b2a9bbad16d1108cc3af6c640","source":{"kind":"arxiv","id":"2502.20847","version":1},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2502.20847","created_at":"2026-07-05T10:21:40Z"},{"alias_kind":"arxiv_version","alias_value":"2502.20847v1","created_at":"2026-07-05T10:21:40Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.20847","created_at":"2026-07-05T10:21:40Z"},{"alias_kind":"pith_short_12","alias_value":"7VGEXAL5VHE5","created_at":"2026-07-05T10:21:40Z"},{"alias_kind":"pith_short_16","alias_value":"7VGEXAL5VHE5PS7T","created_at":"2026-07-05T10:21:40Z"},{"alias_kind":"pith_short_8","alias_value":"7VGEXAL5","created_at":"2026-07-05T10:21:40Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2025:7VGEXAL5VHE5PS7TIGZCX4UFQP","target":"record","payload":{"canonical_record":{"source":{"id":"2502.20847","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-02-28T08:47:03Z","cross_cats_sorted":[],"title_canon_sha256":"f7c0f99c752d9249bb2941d4b6004819fc7151312d96b4f3d41b96b3c5b897e3","abstract_canon_sha256":"6edb87d0bd995b2e3c5539a87105feb77e073fa69e7f31a2cded466df307069e"},"schema_version":"1.0"},"canonical_sha256":"fd4c4b817da9c9d7cbf341b22bf28583d8cff08b2a9bbad16d1108cc3af6c640","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:21:40.895282Z","signature_b64":"OMVcq9d4tD+EP4cU9VDSrcAZpe4y5dIePG/zWpqmEE8G4aAYHN0LiZiccmIJ4XM2awHaSRNCo+0YFTVgFoULCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"fd4c4b817da9c9d7cbf341b22bf28583d8cff08b2a9bbad16d1108cc3af6c640","last_reissued_at":"2026-07-05T10:21:40.894769Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:21:40.894769Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2502.20847","source_version":1,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T10:21:40Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"pRvgS+bCBKZeYqDSIaTd0tzdY+sUR8VCMPFeDGeBPwdOVTdBwc8w7TODcmOUUaWrqmZkU5VSXEcn8KZwrbZYCQ==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-09T00:43:10.882057Z"},"content_sha256":"14a8e5f744f787ff0e6940b5561b8ee5945c3ddb933778aa8e2dc64b658d100e","schema_version":"1.0","event_id":"sha256:14a8e5f744f787ff0e6940b5561b8ee5945c3ddb933778aa8e2dc64b658d100e"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2025:7VGEXAL5VHE5PS7TIGZCX4UFQP","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Gradient Imbalance in Direct Preference Optimization","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Can Jin, Jenq-Neng Hwang, Jingzhe Shi, Lei Li, Qinwei Ma, Serge Belongie","submitted_at":"2025-02-28T08:47:03Z","abstract_excerpt":"Direct Preference Optimization (DPO) has been proposed as a promising alternative to Proximal Policy Optimization (PPO) based Reinforcement Learning with Human Feedback (RLHF). However, empirical evaluations consistently reveal suboptimal performance in DPO compared to common RLHF pipelines. In this work, we conduct a systematic analysis of DPO's training dynamics and identify gradient imbalance as a critical limitation. We demonstrate theoretically and empirically that this imbalance perturbs optimization trajectories, destabilizes learning, and induces suboptimal convergence. To address this"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.20847","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.20847/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T10:21:40Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"STDvwaY8zaCb7CJxQUn/9XvEvkKbgFbcPJrrCfBgprzU/1Rl/DApxcqVCpK340fp57by1VH3Nhjhd0aok+s+DQ==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-09T00:43:10.882546Z"},"content_sha256":"6544eb218d11bb96271bb88fb2350733bcd1035668abf9f21e8185a4f304ac3b","schema_version":"1.0","event_id":"sha256:6544eb218d11bb96271bb88fb2350733bcd1035668abf9f21e8185a4f304ac3b"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/7VGEXAL5VHE5PS7TIGZCX4UFQP/bundle.json","state_url":"https://pith.science/pith/7VGEXAL5VHE5PS7TIGZCX4UFQP/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/7VGEXAL5VHE5PS7TIGZCX4UFQP/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-09T00:43:10Z","links":{"resolver":"https://pith.science/pith/7VGEXAL5VHE5PS7TIGZCX4UFQP","bundle":"https://pith.science/pith/7VGEXAL5VHE5PS7TIGZCX4UFQP/bundle.json","state":"https://pith.science/pith/7VGEXAL5VHE5PS7TIGZCX4UFQP/state.json","well_known_bundle":"https://pith.science/.well-known/pith/7VGEXAL5VHE5PS7TIGZCX4UFQP/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2025:7VGEXAL5VHE5PS7TIGZCX4UFQP","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"6edb87d0bd995b2e3c5539a87105feb77e073fa69e7f31a2cded466df307069e","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-02-28T08:47:03Z","title_canon_sha256":"f7c0f99c752d9249bb2941d4b6004819fc7151312d96b4f3d41b96b3c5b897e3"},"schema_version":"1.0","source":{"id":"2502.20847","kind":"arxiv","version":1}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2502.20847","created_at":"2026-07-05T10:21:40Z"},{"alias_kind":"arxiv_version","alias_value":"2502.20847v1","created_at":"2026-07-05T10:21:40Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.20847","created_at":"2026-07-05T10:21:40Z"},{"alias_kind":"pith_short_12","alias_value":"7VGEXAL5VHE5","created_at":"2026-07-05T10:21:40Z"},{"alias_kind":"pith_short_16","alias_value":"7VGEXAL5VHE5PS7T","created_at":"2026-07-05T10:21:40Z"},{"alias_kind":"pith_short_8","alias_value":"7VGEXAL5","created_at":"2026-07-05T10:21:40Z"}],"graph_snapshots":[{"event_id":"sha256:6544eb218d11bb96271bb88fb2350733bcd1035668abf9f21e8185a4f304ac3b","target":"graph","created_at":"2026-07-05T10:21:40Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2502.20847/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Direct Preference Optimization (DPO) has been proposed as a promising alternative to Proximal Policy Optimization (PPO) based Reinforcement Learning with Human Feedback (RLHF). However, empirical evaluations consistently reveal suboptimal performance in DPO compared to common RLHF pipelines. In this work, we conduct a systematic analysis of DPO's training dynamics and identify gradient imbalance as a critical limitation. We demonstrate theoretically and empirically that this imbalance perturbs optimization trajectories, destabilizes learning, and induces suboptimal convergence. To address this","authors_text":"Can Jin, Jenq-Neng Hwang, Jingzhe Shi, Lei Li, Qinwei Ma, Serge Belongie","cross_cats":[],"headline":"","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-02-28T08:47:03Z","title":"Gradient Imbalance in Direct Preference Optimization"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.20847","kind":"arxiv","version":1},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:14a8e5f744f787ff0e6940b5561b8ee5945c3ddb933778aa8e2dc64b658d100e","target":"record","created_at":"2026-07-05T10:21:40Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"6edb87d0bd995b2e3c5539a87105feb77e073fa69e7f31a2cded466df307069e","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-02-28T08:47:03Z","title_canon_sha256":"f7c0f99c752d9249bb2941d4b6004819fc7151312d96b4f3d41b96b3c5b897e3"},"schema_version":"1.0","source":{"id":"2502.20847","kind":"arxiv","version":1}},"canonical_sha256":"fd4c4b817da9c9d7cbf341b22bf28583d8cff08b2a9bbad16d1108cc3af6c640","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"fd4c4b817da9c9d7cbf341b22bf28583d8cff08b2a9bbad16d1108cc3af6c640","first_computed_at":"2026-07-05T10:21:40.894769Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T10:21:40.894769Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"OMVcq9d4tD+EP4cU9VDSrcAZpe4y5dIePG/zWpqmEE8G4aAYHN0LiZiccmIJ4XM2awHaSRNCo+0YFTVgFoULCw==","signature_status":"signed_v1","signed_at":"2026-07-05T10:21:40.895282Z","signed_message":"canonical_sha256_bytes"},"source_id":"2502.20847","source_kind":"arxiv","source_version":1}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:14a8e5f744f787ff0e6940b5561b8ee5945c3ddb933778aa8e2dc64b658d100e","sha256:6544eb218d11bb96271bb88fb2350733bcd1035668abf9f21e8185a4f304ac3b"],"state_sha256":"bf2025fdfce22be89dbe1e6ac23290314a8e8ca93d16781570de510c5077197b"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"D8DElk3b4tHHNfVyhME47yFy7UiMp8eIxa4oU1rtL91RfImAnNk3cispO2Tso9vp/ebipoYv3DUxPrcaOtqMAQ==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-09T00:43:10.885924Z","bundle_sha256":"520539529b5e96a99d10b238285247b988a427958a8fa3003c123e6d41ce31b1"}}