{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:DEKUWATJUBBGKVCGPKLOWAB5CL","short_pith_number":"pith:DEKUWATJ","schema_version":"1.0","canonical_sha256":"19154b0269a0426554467a96eb003d12cd5de3b8c725a0b59e9ede1ad208bdf2","source":{"kind":"arxiv","id":"2303.03751","version":3},"attestation_state":"computed","paper":{"title":"Zeroth-Order Optimization Meets Human Feedback: Provable Learning via Ranking Oracles","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Dmitry Rybin, Tsung-Hui Chang, Zhiwei Tang","submitted_at":"2023-03-07T09:20:43Z","abstract_excerpt":"In this study, we delve into an emerging optimization challenge involving a black-box objective function that can only be gauged via a ranking oracle-a situation frequently encountered in real-world scenarios, especially when the function is evaluated by human judges. Such challenge is inspired from Reinforcement Learning with Human Feedback (RLHF), an approach recently employed to enhance the performance of Large Language Models (LLMs) using human guidance. We introduce ZO-RankSGD, an innovative zeroth-order optimization algorithm designed to tackle this optimization problem, accompanied by t"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2303.03751","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2023-03-07T09:20:43Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"8f551396bd35e831f2bf4a8fc54a990b1595b06fd558237cdfe5fa7d4cf7cf18","abstract_canon_sha256":"e45a4933686e94520f6aa6e0cb3458842acea2432cc66f3dc0fc9725cc0d5ff3"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:07:26.981487Z","signature_b64":"tev4ySBrvuniKmmC+7/ASVNprw6dGqkEd2n5SBA6e6OTkq9a8nmfP1q98KItvqEKDVYwLgAy8WZEphT3tIQ3DA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"19154b0269a0426554467a96eb003d12cd5de3b8c725a0b59e9ede1ad208bdf2","last_reissued_at":"2026-07-05T08:07:26.981001Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:07:26.981001Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Zeroth-Order Optimization Meets Human Feedback: Provable Learning via Ranking Oracles","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Dmitry Rybin, Tsung-Hui Chang, Zhiwei Tang","submitted_at":"2023-03-07T09:20:43Z","abstract_excerpt":"In this study, we delve into an emerging optimization challenge involving a black-box objective function that can only be gauged via a ranking oracle-a situation frequently encountered in real-world scenarios, especially when the function is evaluated by human judges. Such challenge is inspired from Reinforcement Learning with Human Feedback (RLHF), an approach recently employed to enhance the performance of Large Language Models (LLMs) using human guidance. We introduce ZO-RankSGD, an innovative zeroth-order optimization algorithm designed to tackle this optimization problem, accompanied by t"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2303.03751","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2303.03751/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2303.03751","created_at":"2026-07-05T08:07:26.981060+00:00"},{"alias_kind":"arxiv_version","alias_value":"2303.03751v3","created_at":"2026-07-05T08:07:26.981060+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2303.03751","created_at":"2026-07-05T08:07:26.981060+00:00"},{"alias_kind":"pith_short_12","alias_value":"DEKUWATJUBBG","created_at":"2026-07-05T08:07:26.981060+00:00"},{"alias_kind":"pith_short_16","alias_value":"DEKUWATJUBBGKVCG","created_at":"2026-07-05T08:07:26.981060+00:00"},{"alias_kind":"pith_short_8","alias_value":"DEKUWATJ","created_at":"2026-07-05T08:07:26.981060+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.27082","citing_title":"Finding Stationary Points by Comparisons","ref_index":43,"is_internal_anchor":false},{"citing_arxiv_id":"2606.27305","citing_title":"Sculpting NeRF Geometry: Human-Preference Fine-Tuning of a 3D-Aware Face GAN","ref_index":56,"is_internal_anchor":false},{"citing_arxiv_id":"2604.23540","citing_title":"Oracle Noise: Faster Semantic Spherical Alignment for Interpretable Latent Optimization","ref_index":39,"is_internal_anchor":false},{"citing_arxiv_id":"2605.00650","citing_title":"AdaMeZO: Adam-style Zeroth-Order Optimizer for LLM Fine-tuning Without Maintaining the Moments","ref_index":64,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/DEKUWATJUBBGKVCGPKLOWAB5CL","json":"https://pith.science/pith/DEKUWATJUBBGKVCGPKLOWAB5CL.json","graph_json":"https://pith.science/api/pith-number/DEKUWATJUBBGKVCGPKLOWAB5CL/graph.json","events_json":"https://pith.science/api/pith-number/DEKUWATJUBBGKVCGPKLOWAB5CL/events.json","paper":"https://pith.science/paper/DEKUWATJ"},"agent_actions":{"view_html":"https://pith.science/pith/DEKUWATJUBBGKVCGPKLOWAB5CL","download_json":"https://pith.science/pith/DEKUWATJUBBGKVCGPKLOWAB5CL.json","view_paper":"https://pith.science/paper/DEKUWATJ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2303.03751&json=true","fetch_graph":"https://pith.science/api/pith-number/DEKUWATJUBBGKVCGPKLOWAB5CL/graph.json","fetch_events":"https://pith.science/api/pith-number/DEKUWATJUBBGKVCGPKLOWAB5CL/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/DEKUWATJUBBGKVCGPKLOWAB5CL/action/timestamp_anchor","attest_storage":"https://pith.science/pith/DEKUWATJUBBGKVCGPKLOWAB5CL/action/storage_attestation","attest_author":"https://pith.science/pith/DEKUWATJUBBGKVCGPKLOWAB5CL/action/author_attestation","sign_citation":"https://pith.science/pith/DEKUWATJUBBGKVCGPKLOWAB5CL/action/citation_signature","submit_replication":"https://pith.science/pith/DEKUWATJUBBGKVCGPKLOWAB5CL/action/replication_record"}},"created_at":"2026-07-05T08:07:26.981060+00:00","updated_at":"2026-07-05T08:07:26.981060+00:00"}