{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:RCZODCXTBLSP3MJAXC4523AZHZ","short_pith_number":"pith:RCZODCXT","schema_version":"1.0","canonical_sha256":"88b2e18af30ae4fdb120b8b9dd6c193e58e890f058a3742c84cd1f9bc16b0bbd","source":{"kind":"arxiv","id":"2306.17492","version":2},"attestation_state":"computed","paper":{"title":"Preference Ranking Optimization for Human Alignment","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Bowen Yu, Feifan Song, Fei Huang, Haiyang Yu, Houfeng Wang, Minghao Li, Yongbin Li","submitted_at":"2023-06-30T09:07:37Z","abstract_excerpt":"Large language models (LLMs) often contain misleading content, emphasizing the need to align them with human values to ensure secure AI systems. Reinforcement learning from human feedback (RLHF) has been employed to achieve this alignment. However, it encompasses two main drawbacks: (1) RLHF exhibits complexity, instability, and sensitivity to hyperparameters in contrast to SFT. (2) Despite massive trial-and-error, multiple sampling is reduced to pair-wise contrast, thus lacking contrasts from a macro perspective. In this paper, we propose Preference Ranking Optimization (PRO) as an efficient "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2306.17492","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2023-06-30T09:07:37Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"9466bf07c286bce839bb4794d0c33cd17dfe6b470b9eed6285a19786c7e5081f","abstract_canon_sha256":"70669fff3c425713c50060060e978dffc3c3c6e983fc4160892887114d6e2e40"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:49:55.230234Z","signature_b64":"nOCedPmvIN8Xnbz35noOa7O+0taVLdQSDMjk7miCBxhdQ6PIs4yDMqmDDZ8UCJR5b1KAKnSHzCG33SLYOgI6Dg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"88b2e18af30ae4fdb120b8b9dd6c193e58e890f058a3742c84cd1f9bc16b0bbd","last_reissued_at":"2026-07-05T07:49:55.229771Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:49:55.229771Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Preference Ranking Optimization for Human Alignment","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Bowen Yu, Feifan Song, Fei Huang, Haiyang Yu, Houfeng Wang, Minghao Li, Yongbin Li","submitted_at":"2023-06-30T09:07:37Z","abstract_excerpt":"Large language models (LLMs) often contain misleading content, emphasizing the need to align them with human values to ensure secure AI systems. Reinforcement learning from human feedback (RLHF) has been employed to achieve this alignment. However, it encompasses two main drawbacks: (1) RLHF exhibits complexity, instability, and sensitivity to hyperparameters in contrast to SFT. (2) Despite massive trial-and-error, multiple sampling is reduced to pair-wise contrast, thus lacking contrasts from a macro perspective. In this paper, we propose Preference Ranking Optimization (PRO) as an efficient "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2306.17492","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2306.17492/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2306.17492","created_at":"2026-07-05T07:49:55.229837+00:00"},{"alias_kind":"arxiv_version","alias_value":"2306.17492v2","created_at":"2026-07-05T07:49:55.229837+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2306.17492","created_at":"2026-07-05T07:49:55.229837+00:00"},{"alias_kind":"pith_short_12","alias_value":"RCZODCXTBLSP","created_at":"2026-07-05T07:49:55.229837+00:00"},{"alias_kind":"pith_short_16","alias_value":"RCZODCXTBLSP3MJA","created_at":"2026-07-05T07:49:55.229837+00:00"},{"alias_kind":"pith_short_8","alias_value":"RCZODCXT","created_at":"2026-07-05T07:49:55.229837+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":11,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2402.03300","citing_title":"DeepSeekMath: Pushing the Limits of Mathematical Reasoning in Open Language Models","ref_index":42,"is_internal_anchor":false},{"citing_arxiv_id":"2408.15339","citing_title":"UNA: A Unified Supervised Framework for Efficient LLM Alignment Across Feedback Types","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2409.18169","citing_title":"Harmful Fine-tuning Attacks and Defenses for Large Language Models: A Survey","ref_index":140,"is_internal_anchor":false},{"citing_arxiv_id":"2307.06435","citing_title":"A Comprehensive Overview of Large Language Models","ref_index":171,"is_internal_anchor":false},{"citing_arxiv_id":"2402.13116","citing_title":"A Survey on Knowledge Distillation of Large Language Models","ref_index":90,"is_internal_anchor":false},{"citing_arxiv_id":"2403.07691","citing_title":"ORPO: Monolithic Preference Optimization without Reference Model","ref_index":46,"is_internal_anchor":false},{"citing_arxiv_id":"2304.08244","citing_title":"API-Bank: A Comprehensive Benchmark for Tool-Augmented LLMs","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12288","citing_title":"TokenRatio: Principled Token-Level Preference Optimization via Ratio Matching","ref_index":107,"is_internal_anchor":false},{"citing_arxiv_id":"2308.01825","citing_title":"Scaling Relationship on Learning Mathematical Reasoning with Large Language Models","ref_index":92,"is_internal_anchor":false},{"citing_arxiv_id":"2312.08935","citing_title":"Math-Shepherd: Verify and Reinforce LLMs Step-by-step without Human Annotations","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12288","citing_title":"TokenRatio: Principled Token-Level Preference Optimization via Ratio Matching","ref_index":107,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/RCZODCXTBLSP3MJAXC4523AZHZ","json":"https://pith.science/pith/RCZODCXTBLSP3MJAXC4523AZHZ.json","graph_json":"https://pith.science/api/pith-number/RCZODCXTBLSP3MJAXC4523AZHZ/graph.json","events_json":"https://pith.science/api/pith-number/RCZODCXTBLSP3MJAXC4523AZHZ/events.json","paper":"https://pith.science/paper/RCZODCXT"},"agent_actions":{"view_html":"https://pith.science/pith/RCZODCXTBLSP3MJAXC4523AZHZ","download_json":"https://pith.science/pith/RCZODCXTBLSP3MJAXC4523AZHZ.json","view_paper":"https://pith.science/paper/RCZODCXT","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2306.17492&json=true","fetch_graph":"https://pith.science/api/pith-number/RCZODCXTBLSP3MJAXC4523AZHZ/graph.json","fetch_events":"https://pith.science/api/pith-number/RCZODCXTBLSP3MJAXC4523AZHZ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/RCZODCXTBLSP3MJAXC4523AZHZ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/RCZODCXTBLSP3MJAXC4523AZHZ/action/storage_attestation","attest_author":"https://pith.science/pith/RCZODCXTBLSP3MJAXC4523AZHZ/action/author_attestation","sign_citation":"https://pith.science/pith/RCZODCXTBLSP3MJAXC4523AZHZ/action/citation_signature","submit_replication":"https://pith.science/pith/RCZODCXTBLSP3MJAXC4523AZHZ/action/replication_record"}},"created_at":"2026-07-05T07:49:55.229837+00:00","updated_at":"2026-07-05T07:49:55.229837+00:00"}