{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:ESGHXK7GEVNG7K3AG7BCHASY5R","short_pith_number":"pith:ESGHXK7G","schema_version":"1.0","canonical_sha256":"248c7babe6255a6fab6037c2238258ec68302b8514aacca5e5885ec62e990261","source":{"kind":"arxiv","id":"2407.07880","version":2},"attestation_state":"computed","paper":{"title":"Towards Robust Alignment of Language Models: Distributionally Robustifying Direct Preference Optimization","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Bolin Ding, Jiancan Wu, Jiawei Chen, Jinyang Gao, Junkang Wu, Xiangnan He, Xiang Wang, Yuexiang Xie, Zhengyi Yang","submitted_at":"2024-07-10T17:48:25Z","abstract_excerpt":"This study addresses the challenge of noise in training datasets for Direct Preference Optimization (DPO), a method for aligning Large Language Models (LLMs) with human preferences. We categorize noise into pointwise noise, which includes low-quality data points, and pairwise noise, which encompasses erroneous data pair associations that affect preference rankings. Utilizing Distributionally Robust Optimization (DRO), we enhance DPO's resilience to these types of noise. Our theoretical insights reveal that DPO inherently embeds DRO principles, conferring robustness to pointwise noise, with the"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2407.07880","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-07-10T17:48:25Z","cross_cats_sorted":["cs.AI","cs.CL"],"title_canon_sha256":"122e5d6e569dce5492c15146958a8b734ef8580dc2c2d721877010f4ce2aa7ea","abstract_canon_sha256":"2423439d9175e304549a7874013618654ce6282a2ae07a4678f600831cf70a7d"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:50:42.886149Z","signature_b64":"qmuleK4FV0XE0kAdBmxzMK/S3exM0cjW9VOcXHr+t0cQEuSqjTpm1kGgiKZtUYmZ+EogxrI+3x9OD3WotCw6Bw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"248c7babe6255a6fab6037c2238258ec68302b8514aacca5e5885ec62e990261","last_reissued_at":"2026-07-05T10:50:42.885654Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:50:42.885654Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Towards Robust Alignment of Language Models: Distributionally Robustifying Direct Preference Optimization","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL"],"primary_cat":"cs.LG","authors_text":"Bolin Ding, Jiancan Wu, Jiawei Chen, Jinyang Gao, Junkang Wu, Xiangnan He, Xiang Wang, Yuexiang Xie, Zhengyi Yang","submitted_at":"2024-07-10T17:48:25Z","abstract_excerpt":"This study addresses the challenge of noise in training datasets for Direct Preference Optimization (DPO), a method for aligning Large Language Models (LLMs) with human preferences. We categorize noise into pointwise noise, which includes low-quality data points, and pairwise noise, which encompasses erroneous data pair associations that affect preference rankings. Utilizing Distributionally Robust Optimization (DRO), we enhance DPO's resilience to these types of noise. Our theoretical insights reveal that DPO inherently embeds DRO principles, conferring robustness to pointwise noise, with the"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2407.07880","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2407.07880/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2407.07880","created_at":"2026-07-05T10:50:42.885712+00:00"},{"alias_kind":"arxiv_version","alias_value":"2407.07880v2","created_at":"2026-07-05T10:50:42.885712+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2407.07880","created_at":"2026-07-05T10:50:42.885712+00:00"},{"alias_kind":"pith_short_12","alias_value":"ESGHXK7GEVNG","created_at":"2026-07-05T10:50:42.885712+00:00"},{"alias_kind":"pith_short_16","alias_value":"ESGHXK7GEVNG7K3A","created_at":"2026-07-05T10:50:42.885712+00:00"},{"alias_kind":"pith_short_8","alias_value":"ESGHXK7G","created_at":"2026-07-05T10:50:42.885712+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.09078","citing_title":"The Hidden Bias of Process Reward Models:PRISM for Rewarding the Right Reasoning","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2602.07340","citing_title":"Revisiting Robustness for LLM Safety Alignment via Selective Geometry Control","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11134","citing_title":"Spurious Correlation Learning in Preference Optimization: Mechanisms, Consequences, and Mitigation via Tie Training","ref_index":45,"is_internal_anchor":false},{"citing_arxiv_id":"2605.02495","citing_title":"Efficient Preference Poisoning Attack on Offline RLHF","ref_index":78,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ESGHXK7GEVNG7K3AG7BCHASY5R","json":"https://pith.science/pith/ESGHXK7GEVNG7K3AG7BCHASY5R.json","graph_json":"https://pith.science/api/pith-number/ESGHXK7GEVNG7K3AG7BCHASY5R/graph.json","events_json":"https://pith.science/api/pith-number/ESGHXK7GEVNG7K3AG7BCHASY5R/events.json","paper":"https://pith.science/paper/ESGHXK7G"},"agent_actions":{"view_html":"https://pith.science/pith/ESGHXK7GEVNG7K3AG7BCHASY5R","download_json":"https://pith.science/pith/ESGHXK7GEVNG7K3AG7BCHASY5R.json","view_paper":"https://pith.science/paper/ESGHXK7G","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2407.07880&json=true","fetch_graph":"https://pith.science/api/pith-number/ESGHXK7GEVNG7K3AG7BCHASY5R/graph.json","fetch_events":"https://pith.science/api/pith-number/ESGHXK7GEVNG7K3AG7BCHASY5R/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ESGHXK7GEVNG7K3AG7BCHASY5R/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ESGHXK7GEVNG7K3AG7BCHASY5R/action/storage_attestation","attest_author":"https://pith.science/pith/ESGHXK7GEVNG7K3AG7BCHASY5R/action/author_attestation","sign_citation":"https://pith.science/pith/ESGHXK7GEVNG7K3AG7BCHASY5R/action/citation_signature","submit_replication":"https://pith.science/pith/ESGHXK7GEVNG7K3AG7BCHASY5R/action/replication_record"}},"created_at":"2026-07-05T10:50:42.885712+00:00","updated_at":"2026-07-05T10:50:42.885712+00:00"}