{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:W342R5RWI5G7I7CTOJRORD4JAD","short_pith_number":"pith:W342R5RW","schema_version":"1.0","canonical_sha256":"b6f9a8f636474df47c537262e88f8900c33065f76218db8f789fb1fc3fe2bd4c","source":{"kind":"arxiv","id":"2402.10958","version":2},"attestation_state":"computed","paper":{"title":"Relative Preference Optimization: Enhancing LLM Alignment through Contrasting Responses across Identical and Diverse Prompts","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Hai Huang, Mingyuan Zhou, Weizhu Chen, Yi Gu, Yueqin Yin, Zhendong Wang","submitted_at":"2024-02-12T22:47:57Z","abstract_excerpt":"In the field of large language models (LLMs), aligning models with the diverse preferences of users is a critical challenge. Direct Preference Optimization (DPO) has played a key role in this area. It works by using pairs of preferences derived from the same prompts, and it functions without needing an additional reward model. However, DPO does not fully reflect the complex nature of human learning, which often involves understanding contrasting responses to not only identical but also similar questions. To overcome this shortfall, we propose Relative Preference Optimization (RPO). RPO is desi"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2402.10958","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-02-12T22:47:57Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"f675356a1b9a9efddaf8a468ede2e85c216c8bbdff59e154f194fde8b33eae66","abstract_canon_sha256":"6992facc7f4c9d912bc4b6e591fc251a82939c6457288e30fe640e9ab0c736b0"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:24:02.041331Z","signature_b64":"Ogt3S1NiBtCaiN4TX6VAhLjNt5QB4BpwRbh5bwh0w4TBg5z7VZ27vaVLrbQPQkuxkY08YuZAtYF8AwPsjLGHAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b6f9a8f636474df47c537262e88f8900c33065f76218db8f789fb1fc3fe2bd4c","last_reissued_at":"2026-07-05T08:24:02.040816Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:24:02.040816Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Relative Preference Optimization: Enhancing LLM Alignment through Contrasting Responses across Identical and Diverse Prompts","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Hai Huang, Mingyuan Zhou, Weizhu Chen, Yi Gu, Yueqin Yin, Zhendong Wang","submitted_at":"2024-02-12T22:47:57Z","abstract_excerpt":"In the field of large language models (LLMs), aligning models with the diverse preferences of users is a critical challenge. Direct Preference Optimization (DPO) has played a key role in this area. It works by using pairs of preferences derived from the same prompts, and it functions without needing an additional reward model. However, DPO does not fully reflect the complex nature of human learning, which often involves understanding contrasting responses to not only identical but also similar questions. To overcome this shortfall, we propose Relative Preference Optimization (RPO). RPO is desi"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2402.10958","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2402.10958/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2402.10958","created_at":"2026-07-05T08:24:02.040879+00:00"},{"alias_kind":"arxiv_version","alias_value":"2402.10958v2","created_at":"2026-07-05T08:24:02.040879+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2402.10958","created_at":"2026-07-05T08:24:02.040879+00:00"},{"alias_kind":"pith_short_12","alias_value":"W342R5RWI5G7","created_at":"2026-07-05T08:24:02.040879+00:00"},{"alias_kind":"pith_short_16","alias_value":"W342R5RWI5G7I7CT","created_at":"2026-07-05T08:24:02.040879+00:00"},{"alias_kind":"pith_short_8","alias_value":"W342R5RW","created_at":"2026-07-05T08:24:02.040879+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.28998","citing_title":"Reward-Free Code Alignment from Pretrained or Fine-Tuned LLM: Unpacking the Trade-offs for Code Generation","ref_index":59,"is_internal_anchor":false},{"citing_arxiv_id":"2405.13068","citing_title":"Uncovering Logit Suppression Vulnerabilities in LLM Safety Alignment","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2511.20657","citing_title":"Intelligent Agents with Emotional Intelligence: Current Trends, Challenges, and Future Prospects","ref_index":86,"is_internal_anchor":false},{"citing_arxiv_id":"2602.19974","citing_title":"RL-RIG: A Generative Spatial Reasoner via Intrinsic Reflection","ref_index":51,"is_internal_anchor":false},{"citing_arxiv_id":"2604.24536","citing_title":"Generating Place-Based Compromises Between Two Points of View","ref_index":79,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08037","citing_title":"Beyond Pairs: Your Language Model is Secretly Optimizing a Preference Graph","ref_index":42,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/W342R5RWI5G7I7CTOJRORD4JAD","json":"https://pith.science/pith/W342R5RWI5G7I7CTOJRORD4JAD.json","graph_json":"https://pith.science/api/pith-number/W342R5RWI5G7I7CTOJRORD4JAD/graph.json","events_json":"https://pith.science/api/pith-number/W342R5RWI5G7I7CTOJRORD4JAD/events.json","paper":"https://pith.science/paper/W342R5RW"},"agent_actions":{"view_html":"https://pith.science/pith/W342R5RWI5G7I7CTOJRORD4JAD","download_json":"https://pith.science/pith/W342R5RWI5G7I7CTOJRORD4JAD.json","view_paper":"https://pith.science/paper/W342R5RW","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2402.10958&json=true","fetch_graph":"https://pith.science/api/pith-number/W342R5RWI5G7I7CTOJRORD4JAD/graph.json","fetch_events":"https://pith.science/api/pith-number/W342R5RWI5G7I7CTOJRORD4JAD/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/W342R5RWI5G7I7CTOJRORD4JAD/action/timestamp_anchor","attest_storage":"https://pith.science/pith/W342R5RWI5G7I7CTOJRORD4JAD/action/storage_attestation","attest_author":"https://pith.science/pith/W342R5RWI5G7I7CTOJRORD4JAD/action/author_attestation","sign_citation":"https://pith.science/pith/W342R5RWI5G7I7CTOJRORD4JAD/action/citation_signature","submit_replication":"https://pith.science/pith/W342R5RWI5G7I7CTOJRORD4JAD/action/replication_record"}},"created_at":"2026-07-05T08:24:02.040879+00:00","updated_at":"2026-07-05T08:24:02.040879+00:00"}