{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2024:EMBB3LLIJLSZULK5FRS7B5D6CO","short_pith_number":"pith:EMBB3LLI","canonical_record":{"source":{"id":"2410.18640","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-10-24T11:06:29Z","cross_cats_sorted":[],"title_canon_sha256":"e582b50d5d83be69cfed542de6418c02f6723a395379210237295b50882a59bf","abstract_canon_sha256":"f596f6ff70179c42acd3ad8e11a70045559d4c75b94206115c744f374703d2f1"},"schema_version":"1.0"},"canonical_sha256":"23021dad684ae59a2d5d2c65f0f47e13bcc138393ec1837be61a19805c2f5ec7","source":{"kind":"arxiv","id":"2410.18640","version":2},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2410.18640","created_at":"2026-07-05T10:25:12Z"},{"alias_kind":"arxiv_version","alias_value":"2410.18640v2","created_at":"2026-07-05T10:25:12Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.18640","created_at":"2026-07-05T10:25:12Z"},{"alias_kind":"pith_short_12","alias_value":"EMBB3LLIJLSZ","created_at":"2026-07-05T10:25:12Z"},{"alias_kind":"pith_short_16","alias_value":"EMBB3LLIJLSZULK5","created_at":"2026-07-05T10:25:12Z"},{"alias_kind":"pith_short_8","alias_value":"EMBB3LLI","created_at":"2026-07-05T10:25:12Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2024:EMBB3LLIJLSZULK5FRS7B5D6CO","target":"record","payload":{"canonical_record":{"source":{"id":"2410.18640","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-10-24T11:06:29Z","cross_cats_sorted":[],"title_canon_sha256":"e582b50d5d83be69cfed542de6418c02f6723a395379210237295b50882a59bf","abstract_canon_sha256":"f596f6ff70179c42acd3ad8e11a70045559d4c75b94206115c744f374703d2f1"},"schema_version":"1.0"},"canonical_sha256":"23021dad684ae59a2d5d2c65f0f47e13bcc138393ec1837be61a19805c2f5ec7","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:25:12.646588Z","signature_b64":"fay+y1DM7babr2Q+zLneBd8Bz94pxNwuatSsIuImMF9ZglD/+6b7IGYydKIWeQguFstQi3y6fZHuRGGHPYgMDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"23021dad684ae59a2d5d2c65f0f47e13bcc138393ec1837be61a19805c2f5ec7","last_reissued_at":"2026-07-05T10:25:12.646105Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:25:12.646105Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2410.18640","source_version":2,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T10:25:12Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"cJl/ID0/LDtv5GveUDsKK/xpFI3E3Mnlezeao8/wd9AnJNEchUTLWJCqSvidTT/Jm8k7Xo8IC7MN58o69onzAQ==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-07T14:11:12.467649Z"},"content_sha256":"883f63cfce2f147b39c87036f87b6d6f91928f3d92e7f95a71ff3836e11882f5","schema_version":"1.0","event_id":"sha256:883f63cfce2f147b39c87036f87b6d6f91928f3d92e7f95a71ff3836e11882f5"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2024:EMBB3LLIJLSZULK5FRS7B5D6CO","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Weak-to-Strong Preference Optimization: Stealing Reward from Weak Aligned Model","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Pengfei Liu, Rui Wang, Wenhong Zhu, Xiaofeng Wang, Zhiwei He","submitted_at":"2024-10-24T11:06:29Z","abstract_excerpt":"Aligning language models (LMs) with human preferences has become a key area of research, enabling these models to meet diverse user needs better. Inspired by weak-to-strong generalization, where a strong LM fine-tuned on labels generated by a weaker model can consistently outperform its weak supervisor, we extend this idea to model alignment. In this work, we observe that the alignment behavior in weaker models can be effectively transferred to stronger models and even exhibit an amplification effect. Based on this insight, we propose a method called Weak-to-Strong Preference Optimization (WSP"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.18640","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.18640/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T10:25:12Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"GO7w329WaWT2m1s8UzdyL67dCM8B+bVqEq8L4B4bb4upXT/eEGfJFu8IKa6rOgKeTZW1cqcUqLg1JE0lNH9hAg==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-07T14:11:12.468143Z"},"content_sha256":"0fbb2d42a141676e15a65c53829fe6b203401c125fe3b5a5446baea350bd4e2d","schema_version":"1.0","event_id":"sha256:0fbb2d42a141676e15a65c53829fe6b203401c125fe3b5a5446baea350bd4e2d"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/EMBB3LLIJLSZULK5FRS7B5D6CO/bundle.json","state_url":"https://pith.science/pith/EMBB3LLIJLSZULK5FRS7B5D6CO/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/EMBB3LLIJLSZULK5FRS7B5D6CO/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-07T14:11:12Z","links":{"resolver":"https://pith.science/pith/EMBB3LLIJLSZULK5FRS7B5D6CO","bundle":"https://pith.science/pith/EMBB3LLIJLSZULK5FRS7B5D6CO/bundle.json","state":"https://pith.science/pith/EMBB3LLIJLSZULK5FRS7B5D6CO/state.json","well_known_bundle":"https://pith.science/.well-known/pith/EMBB3LLIJLSZULK5FRS7B5D6CO/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2024:EMBB3LLIJLSZULK5FRS7B5D6CO","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"f596f6ff70179c42acd3ad8e11a70045559d4c75b94206115c744f374703d2f1","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-10-24T11:06:29Z","title_canon_sha256":"e582b50d5d83be69cfed542de6418c02f6723a395379210237295b50882a59bf"},"schema_version":"1.0","source":{"id":"2410.18640","kind":"arxiv","version":2}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2410.18640","created_at":"2026-07-05T10:25:12Z"},{"alias_kind":"arxiv_version","alias_value":"2410.18640v2","created_at":"2026-07-05T10:25:12Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.18640","created_at":"2026-07-05T10:25:12Z"},{"alias_kind":"pith_short_12","alias_value":"EMBB3LLIJLSZ","created_at":"2026-07-05T10:25:12Z"},{"alias_kind":"pith_short_16","alias_value":"EMBB3LLIJLSZULK5","created_at":"2026-07-05T10:25:12Z"},{"alias_kind":"pith_short_8","alias_value":"EMBB3LLI","created_at":"2026-07-05T10:25:12Z"}],"graph_snapshots":[{"event_id":"sha256:0fbb2d42a141676e15a65c53829fe6b203401c125fe3b5a5446baea350bd4e2d","target":"graph","created_at":"2026-07-05T10:25:12Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2410.18640/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Aligning language models (LMs) with human preferences has become a key area of research, enabling these models to meet diverse user needs better. Inspired by weak-to-strong generalization, where a strong LM fine-tuned on labels generated by a weaker model can consistently outperform its weak supervisor, we extend this idea to model alignment. In this work, we observe that the alignment behavior in weaker models can be effectively transferred to stronger models and even exhibit an amplification effect. Based on this insight, we propose a method called Weak-to-Strong Preference Optimization (WSP","authors_text":"Pengfei Liu, Rui Wang, Wenhong Zhu, Xiaofeng Wang, Zhiwei He","cross_cats":[],"headline":"","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-10-24T11:06:29Z","title":"Weak-to-Strong Preference Optimization: Stealing Reward from Weak Aligned Model"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.18640","kind":"arxiv","version":2},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:883f63cfce2f147b39c87036f87b6d6f91928f3d92e7f95a71ff3836e11882f5","target":"record","created_at":"2026-07-05T10:25:12Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"f596f6ff70179c42acd3ad8e11a70045559d4c75b94206115c744f374703d2f1","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-10-24T11:06:29Z","title_canon_sha256":"e582b50d5d83be69cfed542de6418c02f6723a395379210237295b50882a59bf"},"schema_version":"1.0","source":{"id":"2410.18640","kind":"arxiv","version":2}},"canonical_sha256":"23021dad684ae59a2d5d2c65f0f47e13bcc138393ec1837be61a19805c2f5ec7","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"23021dad684ae59a2d5d2c65f0f47e13bcc138393ec1837be61a19805c2f5ec7","first_computed_at":"2026-07-05T10:25:12.646105Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T10:25:12.646105Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"fay+y1DM7babr2Q+zLneBd8Bz94pxNwuatSsIuImMF9ZglD/+6b7IGYydKIWeQguFstQi3y6fZHuRGGHPYgMDA==","signature_status":"signed_v1","signed_at":"2026-07-05T10:25:12.646588Z","signed_message":"canonical_sha256_bytes"},"source_id":"2410.18640","source_kind":"arxiv","source_version":2}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:883f63cfce2f147b39c87036f87b6d6f91928f3d92e7f95a71ff3836e11882f5","sha256:0fbb2d42a141676e15a65c53829fe6b203401c125fe3b5a5446baea350bd4e2d"],"state_sha256":"50cd3900b4a876d20686709773f430924f3f3f82e2943acc2ea72fe0a5ca4712"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"3AWR7FR9xCP5ShkVpepEN9jrzzEMwDkxYtYKXy2Zt3GXuNSw94QMjXyXuHZwtb2GVFe4mjRyU6Rp7+qqafy3CQ==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-07T14:11:12.473357Z","bundle_sha256":"275914d780cfe3e6b90275bad42e92b177ab1d0f3fb10b70e3941ad10ce80640"}}