{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:S4YJM6DB5GAKJSEQM44DDLNQSF","short_pith_number":"pith:S4YJM6DB","schema_version":"1.0","canonical_sha256":"9730967861e980a4c890673831adb09178b63c3c91681b1f06cfd98de96d416c","source":{"kind":"arxiv","id":"2304.05302","version":3},"attestation_state":"computed","paper":{"title":"RRHF: Rank Responses to Align Language Models with Human Feedback without tears","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Chuanqi Tan, Fei Huang, Hongyi Yuan, Songfang Huang, Wei Wang, Zheng Yuan","submitted_at":"2023-04-11T15:53:40Z","abstract_excerpt":"Reinforcement Learning from Human Feedback (RLHF) facilitates the alignment of large language models with human preferences, significantly enhancing the quality of interactions between humans and models. InstructGPT implements RLHF through several stages, including Supervised Fine-Tuning (SFT), reward model training, and Proximal Policy Optimization (PPO). However, PPO is sensitive to hyperparameters and requires multiple models in its standard implementation, making it hard to train and scale up to larger parameter counts. In contrast, we propose a novel learning paradigm called RRHF, which s"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2304.05302","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2023-04-11T15:53:40Z","cross_cats_sorted":[],"title_canon_sha256":"70e2728e70418462d0ff2ebdbc07695529b1af739993c727a3bb89d857ce7d1f","abstract_canon_sha256":"47fbedcf1345809619c1f20df47e4298d460ca1f4df403ad4189e03427e4fc90"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:58:08.990935Z","signature_b64":"0yDXfpw9mqFcD0OJPP8Nd5zRDJ3OlG7yLvy3H4Tu+0k+h/eckjrhE2tH1nSKLmipa5hWCVa/xg5oc5lKCjITDw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"9730967861e980a4c890673831adb09178b63c3c91681b1f06cfd98de96d416c","last_reissued_at":"2026-07-05T06:58:08.990437Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:58:08.990437Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"RRHF: Rank Responses to Align Language Models with Human Feedback without tears","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Chuanqi Tan, Fei Huang, Hongyi Yuan, Songfang Huang, Wei Wang, Zheng Yuan","submitted_at":"2023-04-11T15:53:40Z","abstract_excerpt":"Reinforcement Learning from Human Feedback (RLHF) facilitates the alignment of large language models with human preferences, significantly enhancing the quality of interactions between humans and models. InstructGPT implements RLHF through several stages, including Supervised Fine-Tuning (SFT), reward model training, and Proximal Policy Optimization (PPO). However, PPO is sensitive to hyperparameters and requires multiple models in its standard implementation, making it hard to train and scale up to larger parameter counts. In contrast, we propose a novel learning paradigm called RRHF, which s"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2304.05302","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2304.05302/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2304.05302","created_at":"2026-07-05T06:58:08.990502+00:00"},{"alias_kind":"arxiv_version","alias_value":"2304.05302v3","created_at":"2026-07-05T06:58:08.990502+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2304.05302","created_at":"2026-07-05T06:58:08.990502+00:00"},{"alias_kind":"pith_short_12","alias_value":"S4YJM6DB5GAK","created_at":"2026-07-05T06:58:08.990502+00:00"},{"alias_kind":"pith_short_16","alias_value":"S4YJM6DB5GAKJSEQ","created_at":"2026-07-05T06:58:08.990502+00:00"},{"alias_kind":"pith_short_8","alias_value":"S4YJM6DB","created_at":"2026-07-05T06:58:08.990502+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":27,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.06582","citing_title":"PairAlign: A Framework for Sequence Tokenization via Self-Alignment with Applications to Audio Tokenization","ref_index":59,"is_internal_anchor":false},{"citing_arxiv_id":"2606.28707","citing_title":"BV-Blend: Uncertainty-Weighted Historical Baselines for Stable Critic-Free RL with Verifiable Rewards","ref_index":50,"is_internal_anchor":false},{"citing_arxiv_id":"2605.23244","citing_title":"Convex Optimization for Alignment and Preference Learning on a Single GPU","ref_index":80,"is_internal_anchor":false},{"citing_arxiv_id":"2402.03300","citing_title":"DeepSeekMath: Pushing the Limits of Mathematical Reasoning in Open Language Models","ref_index":58,"is_internal_anchor":false},{"citing_arxiv_id":"2408.15339","citing_title":"UNA: A Unified Supervised Framework for Efficient LLM Alignment Across Feedback Types","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2409.18169","citing_title":"Harmful Fine-tuning Attacks and Defenses for Large Language Models: A Survey","ref_index":175,"is_internal_anchor":false},{"citing_arxiv_id":"2411.15594","citing_title":"A Survey on LLM-as-a-Judge","ref_index":200,"is_internal_anchor":false},{"citing_arxiv_id":"2307.06435","citing_title":"A Comprehensive Overview of Large Language Models","ref_index":170,"is_internal_anchor":false},{"citing_arxiv_id":"2508.15119","citing_title":"Flexible Agent Alignment with Goal Inference from Open-Ended Dialog","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2509.20265","citing_title":"Failure Modes of Maximum Entropy RLHF","ref_index":58,"is_internal_anchor":false},{"citing_arxiv_id":"2510.17881","citing_title":"POPI: Personalizing LLMs via Optimized Natural Language Preference Inference","ref_index":48,"is_internal_anchor":false},{"citing_arxiv_id":"2304.06767","citing_title":"RAFT: Reward rAnked FineTuning for Generative Foundation Model Alignment","ref_index":79,"is_internal_anchor":false},{"citing_arxiv_id":"2402.13116","citing_title":"A Survey on Knowledge Distillation of Large Language Models","ref_index":92,"is_internal_anchor":false},{"citing_arxiv_id":"2308.05374","citing_title":"Trustworthy LLMs: a Survey and Guideline for Evaluating Large Language Models' Alignment","ref_index":39,"is_internal_anchor":false},{"citing_arxiv_id":"2604.09580","citing_title":"OOWM: Structuring Embodied Reasoning and Planning via Object-Oriented Programmatic World Modeling","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12288","citing_title":"TokenRatio: Principled Token-Level Preference Optimization via Ratio Matching","ref_index":108,"is_internal_anchor":false},{"citing_arxiv_id":"2507.21046","citing_title":"A Survey of Self-Evolving Agents: What, When, How, and Where to Evolve on the Path to Artificial Super Intelligence","ref_index":196,"is_internal_anchor":false},{"citing_arxiv_id":"2304.12244","citing_title":"WizardLM: Empowering large pre-trained language models to follow complex instructions","ref_index":49,"is_internal_anchor":false},{"citing_arxiv_id":"2605.04539","citing_title":"RLearner-LLM: Balancing Logical Grounding and Fluency in Large Language Models via Hybrid Direct Preference Optimization","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12288","citing_title":"TokenRatio: Principled Token-Level Preference Optimization via Ratio Matching","ref_index":108,"is_internal_anchor":false},{"citing_arxiv_id":"2605.04539","citing_title":"RLearner-LLM: Balancing Logical Grounding and Fluency in Large Language Models via Hybrid Direct Preference Optimization","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09433","citing_title":"Offline Preference Optimization for Rectified Flow with Noise-Tracked Pairs","ref_index":47,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06582","citing_title":"PairAlign: A Framework for Sequence Tokenization via Self-Alignment with Applications to Audio Tokenization","ref_index":59,"is_internal_anchor":false},{"citing_arxiv_id":"2605.04539","citing_title":"RLearner-LLM: Balancing Logical Grounding and Fluency in Large Language Models via Hybrid Direct Preference Optimization","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2409.19256","citing_title":"HybridFlow: A Flexible and Efficient RLHF Framework","ref_index":97,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/S4YJM6DB5GAKJSEQM44DDLNQSF","json":"https://pith.science/pith/S4YJM6DB5GAKJSEQM44DDLNQSF.json","graph_json":"https://pith.science/api/pith-number/S4YJM6DB5GAKJSEQM44DDLNQSF/graph.json","events_json":"https://pith.science/api/pith-number/S4YJM6DB5GAKJSEQM44DDLNQSF/events.json","paper":"https://pith.science/paper/S4YJM6DB"},"agent_actions":{"view_html":"https://pith.science/pith/S4YJM6DB5GAKJSEQM44DDLNQSF","download_json":"https://pith.science/pith/S4YJM6DB5GAKJSEQM44DDLNQSF.json","view_paper":"https://pith.science/paper/S4YJM6DB","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2304.05302&json=true","fetch_graph":"https://pith.science/api/pith-number/S4YJM6DB5GAKJSEQM44DDLNQSF/graph.json","fetch_events":"https://pith.science/api/pith-number/S4YJM6DB5GAKJSEQM44DDLNQSF/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/S4YJM6DB5GAKJSEQM44DDLNQSF/action/timestamp_anchor","attest_storage":"https://pith.science/pith/S4YJM6DB5GAKJSEQM44DDLNQSF/action/storage_attestation","attest_author":"https://pith.science/pith/S4YJM6DB5GAKJSEQM44DDLNQSF/action/author_attestation","sign_citation":"https://pith.science/pith/S4YJM6DB5GAKJSEQM44DDLNQSF/action/citation_signature","submit_replication":"https://pith.science/pith/S4YJM6DB5GAKJSEQM44DDLNQSF/action/replication_record"}},"created_at":"2026-07-05T06:58:08.990502+00:00","updated_at":"2026-07-05T06:58:08.990502+00:00"}