{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:ZSSK3DQDCJECVIV7SSAKL37ZUJ","short_pith_number":"pith:ZSSK3DQD","schema_version":"1.0","canonical_sha256":"cca4ad8e0312482aa2bf9480a5eff9a273f81a8ce7859e91354e59808a25c741","source":{"kind":"arxiv","id":"2409.17401","version":2},"attestation_state":"computed","paper":{"title":"Zeroth-Order Policy Gradient for Reinforcement Learning from Human Feedback without Reward Inference","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["stat.ML"],"primary_cat":"cs.LG","authors_text":"Lei Ying, Qining Zhang","submitted_at":"2024-09-25T22:20:11Z","abstract_excerpt":"Reward inference (learning a reward model from human preferences) is a critical intermediate step in the Reinforcement Learning from Human Feedback (RLHF) pipeline for fine-tuning Large Language Models (LLMs). In practice, RLHF faces fundamental challenges such as distribution shift, reward model overfitting, and problem misspecification. An alternative approach is direct policy optimization without reward inference, such as Direct Preference Optimization (DPO), which provides a much simpler pipeline and has shown empirical success in LLM applications. However, DPO utilizes the closed-form exp"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2409.17401","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-09-25T22:20:11Z","cross_cats_sorted":["stat.ML"],"title_canon_sha256":"8cf247e983ecd46481b57ca31c0098398b04d99af8815c0b81414ea9f432564f","abstract_canon_sha256":"dde5d7d76550560c3fcb9a1e1c4b22fc0cec852419d3e5735a6e500909fbb389"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:22:50.620926Z","signature_b64":"+kCq/JuIarEsQXEtn7/QI97uEgtzLQioaIrLKfHQIlkopchCiPI8zTChPdIuRdSLQcAtRteBvDlrVnRFO+tEDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"cca4ad8e0312482aa2bf9480a5eff9a273f81a8ce7859e91354e59808a25c741","last_reissued_at":"2026-07-05T10:22:50.620186Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:22:50.620186Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Zeroth-Order Policy Gradient for Reinforcement Learning from Human Feedback without Reward Inference","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["stat.ML"],"primary_cat":"cs.LG","authors_text":"Lei Ying, Qining Zhang","submitted_at":"2024-09-25T22:20:11Z","abstract_excerpt":"Reward inference (learning a reward model from human preferences) is a critical intermediate step in the Reinforcement Learning from Human Feedback (RLHF) pipeline for fine-tuning Large Language Models (LLMs). In practice, RLHF faces fundamental challenges such as distribution shift, reward model overfitting, and problem misspecification. An alternative approach is direct policy optimization without reward inference, such as Direct Preference Optimization (DPO), which provides a much simpler pipeline and has shown empirical success in LLM applications. However, DPO utilizes the closed-form exp"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2409.17401","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2409.17401/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2409.17401","created_at":"2026-07-05T10:22:50.620286+00:00"},{"alias_kind":"arxiv_version","alias_value":"2409.17401v2","created_at":"2026-07-05T10:22:50.620286+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2409.17401","created_at":"2026-07-05T10:22:50.620286+00:00"},{"alias_kind":"pith_short_12","alias_value":"ZSSK3DQDCJEC","created_at":"2026-07-05T10:22:50.620286+00:00"},{"alias_kind":"pith_short_16","alias_value":"ZSSK3DQDCJECVIV7","created_at":"2026-07-05T10:22:50.620286+00:00"},{"alias_kind":"pith_short_8","alias_value":"ZSSK3DQD","created_at":"2026-07-05T10:22:50.620286+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.14970","citing_title":"Zero-order Parameter-free Optimization for LMO-based Methods: Novel Approach for Efficient Fine-tuning","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2605.15697","citing_title":"Distributed Zeroth-Order Policy Gradient for Networked Multi-agent Reinforcement Learning from Human Feedback","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2604.19024","citing_title":"Policy Gradient Primal-Dual Method for Safe Reinforcement Learning from Human Feedback","ref_index":11,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ZSSK3DQDCJECVIV7SSAKL37ZUJ","json":"https://pith.science/pith/ZSSK3DQDCJECVIV7SSAKL37ZUJ.json","graph_json":"https://pith.science/api/pith-number/ZSSK3DQDCJECVIV7SSAKL37ZUJ/graph.json","events_json":"https://pith.science/api/pith-number/ZSSK3DQDCJECVIV7SSAKL37ZUJ/events.json","paper":"https://pith.science/paper/ZSSK3DQD"},"agent_actions":{"view_html":"https://pith.science/pith/ZSSK3DQDCJECVIV7SSAKL37ZUJ","download_json":"https://pith.science/pith/ZSSK3DQDCJECVIV7SSAKL37ZUJ.json","view_paper":"https://pith.science/paper/ZSSK3DQD","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2409.17401&json=true","fetch_graph":"https://pith.science/api/pith-number/ZSSK3DQDCJECVIV7SSAKL37ZUJ/graph.json","fetch_events":"https://pith.science/api/pith-number/ZSSK3DQDCJECVIV7SSAKL37ZUJ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ZSSK3DQDCJECVIV7SSAKL37ZUJ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ZSSK3DQDCJECVIV7SSAKL37ZUJ/action/storage_attestation","attest_author":"https://pith.science/pith/ZSSK3DQDCJECVIV7SSAKL37ZUJ/action/author_attestation","sign_citation":"https://pith.science/pith/ZSSK3DQDCJECVIV7SSAKL37ZUJ/action/citation_signature","submit_replication":"https://pith.science/pith/ZSSK3DQDCJECVIV7SSAKL37ZUJ/action/replication_record"}},"created_at":"2026-07-05T10:22:50.620286+00:00","updated_at":"2026-07-05T10:22:50.620286+00:00"}