{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:OPDVS7JXUFSKSOGKBJGGBFQ2GT","short_pith_number":"pith:OPDVS7JX","schema_version":"1.0","canonical_sha256":"73c7597d37a164a938ca0a4c60961a34f658b9821a4897b9e227934ac101de55","source":{"kind":"arxiv","id":"2405.14103","version":1},"attestation_state":"computed","paper":{"title":"Online Self-Preferring Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Bo Ding, Cheng Yang, Dawei Feng, Hanyang Peng, Huaimin Wang, Kele Xu, Yuanzhao Zhai, Yue Yu, Zhuo Zhang","submitted_at":"2024-05-23T02:13:34Z","abstract_excerpt":"Aligning with human preference datasets has been critical to the success of large language models (LLMs). Reinforcement learning from human feedback (RLHF) employs a costly reward model to provide feedback for on-policy sampling responses. Recently, offline methods that directly fit responses with binary preferences in the dataset have emerged as alternatives. However, existing methods do not explicitly model preference strength information, which is crucial for distinguishing different response pairs. To overcome this limitation, we propose Online Self-Preferring (OSP) language models to lear"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2405.14103","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-05-23T02:13:34Z","cross_cats_sorted":[],"title_canon_sha256":"5d7ab8946efb5195d5287b48c40733b0332413dfca023fc8d93a9afc8dbc20a7","abstract_canon_sha256":"14684b30abee4f4242518634386b6c79409efffd842f252704d826edd3d298d2"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:22:10.327515Z","signature_b64":"LROpoxf/gJ33kdS0fr5e9zp9reVlhXrBAsmHrdSqUxePWqytKPaDs5x3H532VhD/sSQvjN6NWvZQoiQCE8XMAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"73c7597d37a164a938ca0a4c60961a34f658b9821a4897b9e227934ac101de55","last_reissued_at":"2026-07-05T08:22:10.327041Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:22:10.327041Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Online Self-Preferring Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Bo Ding, Cheng Yang, Dawei Feng, Hanyang Peng, Huaimin Wang, Kele Xu, Yuanzhao Zhai, Yue Yu, Zhuo Zhang","submitted_at":"2024-05-23T02:13:34Z","abstract_excerpt":"Aligning with human preference datasets has been critical to the success of large language models (LLMs). Reinforcement learning from human feedback (RLHF) employs a costly reward model to provide feedback for on-policy sampling responses. Recently, offline methods that directly fit responses with binary preferences in the dataset have emerged as alternatives. However, existing methods do not explicitly model preference strength information, which is crucial for distinguishing different response pairs. To overcome this limitation, we propose Online Self-Preferring (OSP) language models to lear"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2405.14103","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2405.14103/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2405.14103","created_at":"2026-07-05T08:22:10.327097+00:00"},{"alias_kind":"arxiv_version","alias_value":"2405.14103v1","created_at":"2026-07-05T08:22:10.327097+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2405.14103","created_at":"2026-07-05T08:22:10.327097+00:00"},{"alias_kind":"pith_short_12","alias_value":"OPDVS7JXUFSK","created_at":"2026-07-05T08:22:10.327097+00:00"},{"alias_kind":"pith_short_16","alias_value":"OPDVS7JXUFSKSOGK","created_at":"2026-07-05T08:22:10.327097+00:00"},{"alias_kind":"pith_short_8","alias_value":"OPDVS7JX","created_at":"2026-07-05T08:22:10.327097+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/OPDVS7JXUFSKSOGKBJGGBFQ2GT","json":"https://pith.science/pith/OPDVS7JXUFSKSOGKBJGGBFQ2GT.json","graph_json":"https://pith.science/api/pith-number/OPDVS7JXUFSKSOGKBJGGBFQ2GT/graph.json","events_json":"https://pith.science/api/pith-number/OPDVS7JXUFSKSOGKBJGGBFQ2GT/events.json","paper":"https://pith.science/paper/OPDVS7JX"},"agent_actions":{"view_html":"https://pith.science/pith/OPDVS7JXUFSKSOGKBJGGBFQ2GT","download_json":"https://pith.science/pith/OPDVS7JXUFSKSOGKBJGGBFQ2GT.json","view_paper":"https://pith.science/paper/OPDVS7JX","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2405.14103&json=true","fetch_graph":"https://pith.science/api/pith-number/OPDVS7JXUFSKSOGKBJGGBFQ2GT/graph.json","fetch_events":"https://pith.science/api/pith-number/OPDVS7JXUFSKSOGKBJGGBFQ2GT/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/OPDVS7JXUFSKSOGKBJGGBFQ2GT/action/timestamp_anchor","attest_storage":"https://pith.science/pith/OPDVS7JXUFSKSOGKBJGGBFQ2GT/action/storage_attestation","attest_author":"https://pith.science/pith/OPDVS7JXUFSKSOGKBJGGBFQ2GT/action/author_attestation","sign_citation":"https://pith.science/pith/OPDVS7JXUFSKSOGKBJGGBFQ2GT/action/citation_signature","submit_replication":"https://pith.science/pith/OPDVS7JXUFSKSOGKBJGGBFQ2GT/action/replication_record"}},"created_at":"2026-07-05T08:22:10.327097+00:00","updated_at":"2026-07-05T08:22:10.327097+00:00"}