{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:XOCMAYAGECYLHUW6GUYKB3JPZD","short_pith_number":"pith:XOCMAYAG","schema_version":"1.0","canonical_sha256":"bb84c0600620b0b3d2de3530a0ed2fc8e8fadbb6fd011898231ea969898508f6","source":{"kind":"arxiv","id":"2405.20335","version":1},"attestation_state":"computed","paper":{"title":"Xwin-LM: Strong and Scalable Alignment Practice for LLMs","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Bolin Ni, Gaofeng Meng, Han Hu, Houwen Peng, Jingcheng Hu, Yixuan Wei, Zheng Zhang","submitted_at":"2024-05-30T17:59:31Z","abstract_excerpt":"In this work, we present Xwin-LM, a comprehensive suite of alignment methodologies for large language models (LLMs). This suite encompasses several key techniques, including supervised finetuning (SFT), reward modeling (RM), rejection sampling finetuning (RS), and direct preference optimization (DPO). The key components are as follows: (1) Xwin-LM-SFT, models initially finetuned with high-quality instruction data; (2) Xwin-Pair, a large-scale, multi-turn preference dataset meticulously annotated using GPT-4; (3) Xwin-RM, reward models trained on Xwin-Pair, developed at scales of 7B, 13B, and 7"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2405.20335","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-05-30T17:59:31Z","cross_cats_sorted":[],"title_canon_sha256":"43ebd24eac0bfceb3682caf74de0c36bd72f23cd9e007a5325a42100276efe81","abstract_canon_sha256":"af575fa5b4885f7da8b8ee33052ba23640ff8665d7626f38e4e1fbd79b633970"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:25:24.449256Z","signature_b64":"osRZXHvbVj2BvO8l6vT+0u9GrI4aEn6tE4XCZ+orsJyiY7d2dl9fHlmCuSz9qrO6cRN3kNje0NJGumPM3XdKDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"bb84c0600620b0b3d2de3530a0ed2fc8e8fadbb6fd011898231ea969898508f6","last_reissued_at":"2026-07-05T08:25:24.448785Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:25:24.448785Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Xwin-LM: Strong and Scalable Alignment Practice for LLMs","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Bolin Ni, Gaofeng Meng, Han Hu, Houwen Peng, Jingcheng Hu, Yixuan Wei, Zheng Zhang","submitted_at":"2024-05-30T17:59:31Z","abstract_excerpt":"In this work, we present Xwin-LM, a comprehensive suite of alignment methodologies for large language models (LLMs). This suite encompasses several key techniques, including supervised finetuning (SFT), reward modeling (RM), rejection sampling finetuning (RS), and direct preference optimization (DPO). The key components are as follows: (1) Xwin-LM-SFT, models initially finetuned with high-quality instruction data; (2) Xwin-Pair, a large-scale, multi-turn preference dataset meticulously annotated using GPT-4; (3) Xwin-RM, reward models trained on Xwin-Pair, developed at scales of 7B, 13B, and 7"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2405.20335","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2405.20335/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2405.20335","created_at":"2026-07-05T08:25:24.448849+00:00"},{"alias_kind":"arxiv_version","alias_value":"2405.20335v1","created_at":"2026-07-05T08:25:24.448849+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2405.20335","created_at":"2026-07-05T08:25:24.448849+00:00"},{"alias_kind":"pith_short_12","alias_value":"XOCMAYAGECYL","created_at":"2026-07-05T08:25:24.448849+00:00"},{"alias_kind":"pith_short_16","alias_value":"XOCMAYAGECYLHUW6","created_at":"2026-07-05T08:25:24.448849+00:00"},{"alias_kind":"pith_short_8","alias_value":"XOCMAYAG","created_at":"2026-07-05T08:25:24.448849+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2502.02871","citing_title":"Position: Multimodal Large Language Models Can Significantly Advance Scientific Reasoning","ref_index":139,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/XOCMAYAGECYLHUW6GUYKB3JPZD","json":"https://pith.science/pith/XOCMAYAGECYLHUW6GUYKB3JPZD.json","graph_json":"https://pith.science/api/pith-number/XOCMAYAGECYLHUW6GUYKB3JPZD/graph.json","events_json":"https://pith.science/api/pith-number/XOCMAYAGECYLHUW6GUYKB3JPZD/events.json","paper":"https://pith.science/paper/XOCMAYAG"},"agent_actions":{"view_html":"https://pith.science/pith/XOCMAYAGECYLHUW6GUYKB3JPZD","download_json":"https://pith.science/pith/XOCMAYAGECYLHUW6GUYKB3JPZD.json","view_paper":"https://pith.science/paper/XOCMAYAG","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2405.20335&json=true","fetch_graph":"https://pith.science/api/pith-number/XOCMAYAGECYLHUW6GUYKB3JPZD/graph.json","fetch_events":"https://pith.science/api/pith-number/XOCMAYAGECYLHUW6GUYKB3JPZD/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/XOCMAYAGECYLHUW6GUYKB3JPZD/action/timestamp_anchor","attest_storage":"https://pith.science/pith/XOCMAYAGECYLHUW6GUYKB3JPZD/action/storage_attestation","attest_author":"https://pith.science/pith/XOCMAYAGECYLHUW6GUYKB3JPZD/action/author_attestation","sign_citation":"https://pith.science/pith/XOCMAYAGECYLHUW6GUYKB3JPZD/action/citation_signature","submit_replication":"https://pith.science/pith/XOCMAYAGECYLHUW6GUYKB3JPZD/action/replication_record"}},"created_at":"2026-07-05T08:25:24.448849+00:00","updated_at":"2026-07-05T08:25:24.448849+00:00"}