{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:4XPIADPHVG7MOA4DBHZH4IW6N3","short_pith_number":"pith:4XPIADPH","schema_version":"1.0","canonical_sha256":"e5de800de7a9bec7038309f27e22de6efe3994f11501fa34844f70218996e485","source":{"kind":"arxiv","id":"2504.07856","version":3},"attestation_state":"computed","paper":{"title":"2D-Curri-DPO: Two-Dimensional Curriculum Learning for Direct Preference Optimization","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Mengyang Li, Zhong Zhang","submitted_at":"2025-04-10T15:32:00Z","abstract_excerpt":"Aligning large language models with human preferences is crucial for their safe deployment. While Direct Preference Optimization (DPO) offers an efficient alternative to reinforcement learning from human feedback, traditional DPO methods are limited by their reliance on single preference pairs. Recent work like Curriculum-DPO integrates multiple pairs using a one-dimensional difficulty curriculum based on pairwise distinguishability (PD), but overlooks the complexity of the input prompt itself. To address this, we propose 2D-Curri-DPO, a novel framework employing a two-dimensional curriculum t"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2504.07856","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2025-04-10T15:32:00Z","cross_cats_sorted":[],"title_canon_sha256":"12c04a50f9d9ad7bddd82583df969420a0f0c4c69d88274c17d2e9ce1ef01881","abstract_canon_sha256":"4270507560a4329935ab2ff6740e280b19d9d6b19ea57750dbe91eed149d1c53"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:44:47.659872Z","signature_b64":"eJ6+y/wsio0RrUtUlg78HOBCpmyo1Kk2WxQCV0uP/NJjD0TKdtK16k7k6FdyAQbySqT7QetPBeHESoLrN40gDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"e5de800de7a9bec7038309f27e22de6efe3994f11501fa34844f70218996e485","last_reissued_at":"2026-07-05T11:44:47.659395Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:44:47.659395Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"2D-Curri-DPO: Two-Dimensional Curriculum Learning for Direct Preference Optimization","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Mengyang Li, Zhong Zhang","submitted_at":"2025-04-10T15:32:00Z","abstract_excerpt":"Aligning large language models with human preferences is crucial for their safe deployment. While Direct Preference Optimization (DPO) offers an efficient alternative to reinforcement learning from human feedback, traditional DPO methods are limited by their reliance on single preference pairs. Recent work like Curriculum-DPO integrates multiple pairs using a one-dimensional difficulty curriculum based on pairwise distinguishability (PD), but overlooks the complexity of the input prompt itself. To address this, we propose 2D-Curri-DPO, a novel framework employing a two-dimensional curriculum t"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2504.07856","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2504.07856/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2504.07856","created_at":"2026-07-05T11:44:47.659454+00:00"},{"alias_kind":"arxiv_version","alias_value":"2504.07856v3","created_at":"2026-07-05T11:44:47.659454+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2504.07856","created_at":"2026-07-05T11:44:47.659454+00:00"},{"alias_kind":"pith_short_12","alias_value":"4XPIADPHVG7M","created_at":"2026-07-05T11:44:47.659454+00:00"},{"alias_kind":"pith_short_16","alias_value":"4XPIADPHVG7MOA4D","created_at":"2026-07-05T11:44:47.659454+00:00"},{"alias_kind":"pith_short_8","alias_value":"4XPIADPH","created_at":"2026-07-05T11:44:47.659454+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":2,"sample":[{"citing_arxiv_id":"2605.11679","citing_title":"Explaining and Breaking the Safety-Helpfulness Ceiling via Preference Dimensional Expansion","ref_index":24,"is_internal_anchor":true},{"citing_arxiv_id":"2605.11679","citing_title":"Explaining and Breaking the Safety-Helpfulness Ceiling via Preference Dimensional Expansion","ref_index":24,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/4XPIADPHVG7MOA4DBHZH4IW6N3","json":"https://pith.science/pith/4XPIADPHVG7MOA4DBHZH4IW6N3.json","graph_json":"https://pith.science/api/pith-number/4XPIADPHVG7MOA4DBHZH4IW6N3/graph.json","events_json":"https://pith.science/api/pith-number/4XPIADPHVG7MOA4DBHZH4IW6N3/events.json","paper":"https://pith.science/paper/4XPIADPH"},"agent_actions":{"view_html":"https://pith.science/pith/4XPIADPHVG7MOA4DBHZH4IW6N3","download_json":"https://pith.science/pith/4XPIADPHVG7MOA4DBHZH4IW6N3.json","view_paper":"https://pith.science/paper/4XPIADPH","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2504.07856&json=true","fetch_graph":"https://pith.science/api/pith-number/4XPIADPHVG7MOA4DBHZH4IW6N3/graph.json","fetch_events":"https://pith.science/api/pith-number/4XPIADPHVG7MOA4DBHZH4IW6N3/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/4XPIADPHVG7MOA4DBHZH4IW6N3/action/timestamp_anchor","attest_storage":"https://pith.science/pith/4XPIADPHVG7MOA4DBHZH4IW6N3/action/storage_attestation","attest_author":"https://pith.science/pith/4XPIADPHVG7MOA4DBHZH4IW6N3/action/author_attestation","sign_citation":"https://pith.science/pith/4XPIADPHVG7MOA4DBHZH4IW6N3/action/citation_signature","submit_replication":"https://pith.science/pith/4XPIADPHVG7MOA4DBHZH4IW6N3/action/replication_record"}},"created_at":"2026-07-05T11:44:47.659454+00:00","updated_at":"2026-07-05T11:44:47.659454+00:00"}