{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:NI4ABZHOT2NFHOKLOA5CWPNJUN","short_pith_number":"pith:NI4ABZHO","schema_version":"1.0","canonical_sha256":"6a3800e4ee9e9a53b94b703a2b3da9a37cbdf5ccd5490a053b65fc69c8206212","source":{"kind":"arxiv","id":"2405.16681","version":2},"attestation_state":"computed","paper":{"title":"Triple Preference Optimization: Achieving Better Alignment using a Single Step Optimization","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Amir Saeidi, Aswin RRV, Chitta Baral, Kashif Rasul, Shivanshu Verma","submitted_at":"2024-05-26T20:18:11Z","abstract_excerpt":"Reinforcement Learning with Human Feedback (RLHF) enhances the alignment of Large Language Models (LLMs). However, its limitations have led to the development of Direct Preference Optimization (DPO), an RL-free approach designed to overcome these shortcomings. While studies have shown that DPO improves instruction-following capabilities, it negatively impacts the reasoning ability of LLMs. Additionally, DPO is highly sensitive to judgment noise in preference datasets and the size of the training set. Although several modifications to DPO have been proposed, they still fail to fully resolve the"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2405.16681","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-05-26T20:18:11Z","cross_cats_sorted":[],"title_canon_sha256":"4a1334f8fee055fd5c533b943555cee44c20c9ee4b8ef2fcf3e64c055f8ab4d0","abstract_canon_sha256":"bc1e2267d50b4447263b7d17cf1d8b5a6a18856f41287abcdb952617b52cc880"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:15:43.783101Z","signature_b64":"MTQnopAGKSkYzGzojix9079jpxC7jmTdJa80kiY+U2HVjaQr+IaiIDp9JDQkKJHLqEIKmz6QOjujEAlmfiTdDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"6a3800e4ee9e9a53b94b703a2b3da9a37cbdf5ccd5490a053b65fc69c8206212","last_reissued_at":"2026-07-05T10:15:43.782543Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:15:43.782543Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Triple Preference Optimization: Achieving Better Alignment using a Single Step Optimization","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Amir Saeidi, Aswin RRV, Chitta Baral, Kashif Rasul, Shivanshu Verma","submitted_at":"2024-05-26T20:18:11Z","abstract_excerpt":"Reinforcement Learning with Human Feedback (RLHF) enhances the alignment of Large Language Models (LLMs). However, its limitations have led to the development of Direct Preference Optimization (DPO), an RL-free approach designed to overcome these shortcomings. While studies have shown that DPO improves instruction-following capabilities, it negatively impacts the reasoning ability of LLMs. Additionally, DPO is highly sensitive to judgment noise in preference datasets and the size of the training set. Although several modifications to DPO have been proposed, they still fail to fully resolve the"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2405.16681","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2405.16681/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2405.16681","created_at":"2026-07-05T10:15:43.782614+00:00"},{"alias_kind":"arxiv_version","alias_value":"2405.16681v2","created_at":"2026-07-05T10:15:43.782614+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2405.16681","created_at":"2026-07-05T10:15:43.782614+00:00"},{"alias_kind":"pith_short_12","alias_value":"NI4ABZHOT2NF","created_at":"2026-07-05T10:15:43.782614+00:00"},{"alias_kind":"pith_short_16","alias_value":"NI4ABZHOT2NFHOKL","created_at":"2026-07-05T10:15:43.782614+00:00"},{"alias_kind":"pith_short_8","alias_value":"NI4ABZHO","created_at":"2026-07-05T10:15:43.782614+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.08472","citing_title":"Mid-Training with Self-Generated Data Improves Reinforcement Learning in Language Models","ref_index":43,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/NI4ABZHOT2NFHOKLOA5CWPNJUN","json":"https://pith.science/pith/NI4ABZHOT2NFHOKLOA5CWPNJUN.json","graph_json":"https://pith.science/api/pith-number/NI4ABZHOT2NFHOKLOA5CWPNJUN/graph.json","events_json":"https://pith.science/api/pith-number/NI4ABZHOT2NFHOKLOA5CWPNJUN/events.json","paper":"https://pith.science/paper/NI4ABZHO"},"agent_actions":{"view_html":"https://pith.science/pith/NI4ABZHOT2NFHOKLOA5CWPNJUN","download_json":"https://pith.science/pith/NI4ABZHOT2NFHOKLOA5CWPNJUN.json","view_paper":"https://pith.science/paper/NI4ABZHO","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2405.16681&json=true","fetch_graph":"https://pith.science/api/pith-number/NI4ABZHOT2NFHOKLOA5CWPNJUN/graph.json","fetch_events":"https://pith.science/api/pith-number/NI4ABZHOT2NFHOKLOA5CWPNJUN/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/NI4ABZHOT2NFHOKLOA5CWPNJUN/action/timestamp_anchor","attest_storage":"https://pith.science/pith/NI4ABZHOT2NFHOKLOA5CWPNJUN/action/storage_attestation","attest_author":"https://pith.science/pith/NI4ABZHOT2NFHOKLOA5CWPNJUN/action/author_attestation","sign_citation":"https://pith.science/pith/NI4ABZHOT2NFHOKLOA5CWPNJUN/action/citation_signature","submit_replication":"https://pith.science/pith/NI4ABZHOT2NFHOKLOA5CWPNJUN/action/replication_record"}},"created_at":"2026-07-05T10:15:43.782614+00:00","updated_at":"2026-07-05T10:15:43.782614+00:00"}