{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2026:7Q7GUWLHTTC46OZIG2RUPN3ERS","short_pith_number":"pith:7Q7GUWLH","canonical_record":{"source":{"id":"2604.18239","kind":"arxiv","version":4},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2026-04-20T13:23:27Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"ed324f6d54726228edb39bb3947162d0bdd4409f98aff613834639a5f762f8b9","abstract_canon_sha256":"bb641315dd1ec6ab5e83a45584826fbc89fded5f76ada51774d65df556bde155"},"schema_version":"1.0"},"canonical_sha256":"fc3e6a59679cc5cf3b2836a347b7648c809cb6022e016e42d3c89538445328b8","source":{"kind":"arxiv","id":"2604.18239","version":4},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2604.18239","created_at":"2026-07-24T01:24:11Z"},{"alias_kind":"arxiv_version","alias_value":"2604.18239v4","created_at":"2026-07-24T01:24:11Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2604.18239","created_at":"2026-07-24T01:24:11Z"},{"alias_kind":"pith_short_12","alias_value":"7Q7GUWLHTTC4","created_at":"2026-07-24T01:24:11Z"},{"alias_kind":"pith_short_16","alias_value":"7Q7GUWLHTTC46OZI","created_at":"2026-07-24T01:24:11Z"},{"alias_kind":"pith_short_8","alias_value":"7Q7GUWLH","created_at":"2026-07-24T01:24:11Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2026:7Q7GUWLHTTC46OZIG2RUPN3ERS","target":"record","payload":{"canonical_record":{"source":{"id":"2604.18239","kind":"arxiv","version":4},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2026-04-20T13:23:27Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"ed324f6d54726228edb39bb3947162d0bdd4409f98aff613834639a5f762f8b9","abstract_canon_sha256":"bb641315dd1ec6ab5e83a45584826fbc89fded5f76ada51774d65df556bde155"},"schema_version":"1.0"},"canonical_sha256":"fc3e6a59679cc5cf3b2836a347b7648c809cb6022e016e42d3c89538445328b8","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-24T01:24:11.399258Z","signature_b64":"5GINvnQmkvNxU6/ZiXrX3KmV4tl1ZN+RFDgKx2AiOFFSgrNH8J7zy/U6L9m9tgDFmmI5JVJERjFTGBskobf+Dg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"fc3e6a59679cc5cf3b2836a347b7648c809cb6022e016e42d3c89538445328b8","last_reissued_at":"2026-07-24T01:24:11.398310Z","signature_status":"signed_v1","first_computed_at":"2026-07-24T01:24:11.398310Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2604.18239","source_version":4,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-24T01:24:11Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"mv7RMqvYX6C2efVG8aTllAOVzXxEAs1nsvfMRp441frmiS+g9TShAF7akjnDu599qTGUbDaknGiBVq7ZWUNXBw==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-12T21:04:33.202918Z"},"content_sha256":"9037f5855c7b707cd24c9874cc1959835ad5567ec5790e2c0dbeed9fe643115e","schema_version":"1.0","event_id":"sha256:9037f5855c7b707cd24c9874cc1959835ad5567ec5790e2c0dbeed9fe643115e"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2026:7Q7GUWLHTTC46OZIG2RUPN3ERS","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Towards Disentangled Preference Optimization Dynamics: Suppress the Loser, Preserve the Winner","license":"http://creativecommons.org/licenses/by/4.0/","headline":"Preference optimization can suppress rejected responses without harming chosen ones by satisfying the disentanglement band condition.","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Delu Zeng, John Paisley, Junmei Yang, Min Chen, Qibin Zhao, Wei Chen, Yubing Wu, Zhou Wang","submitted_at":"2026-04-20T13:23:27Z","abstract_excerpt":"Preference optimization is widely used to align large language models (LLMs) with human preferences. However, many margin-based methods also suppress the chosen response when they try to suppress the rejected one, and there is no general way to prevent this across different objectives.\n  We address this issue with a unified incentive-score decomposition of preference optimization, revealing that different objectives share the same local update directions and differ only in their scalar weights.\n  This decomposition provides a common framework for analyzing objectives that were previously studi"},"claims":{"count":4,"items":[{"kind":"strongest_claim","text":"Using the DB, we propose reward calibration (RC), a plug-and-play method that adaptively rebalances the updates for chosen and rejected responses to satisfy the DB, without redesigning the base objective. Empirical results show that RC leads to more disentangled dynamics, with better downstream performance observed across several settings.","source":"verdict.strongest_claim","status":"machine_extracted","claim_id":"C1","attestation":"unclaimed"},{"kind":"weakest_assumption","text":"The decomposition assumes that different objectives share the same local update directions and differ only in scalar weights, which must hold for the disentanglement band analysis to apply across methods.","source":"verdict.weakest_assumption","status":"machine_extracted","claim_id":"C2","attestation":"unclaimed"},{"kind":"one_line_summary","text":"A unified incentive-score decomposition of preference optimization reveals the disentanglement band condition and reward calibration method that enables suppressing losers while preserving winners in LLM training.","source":"verdict.one_line_summary","status":"machine_extracted","claim_id":"C3","attestation":"unclaimed"},{"kind":"headline","text":"Preference optimization can suppress rejected responses without harming chosen ones by satisfying the disentanglement band condition.","source":"verdict.pith_extraction.headline","status":"machine_extracted","claim_id":"C4","attestation":"unclaimed"}],"snapshot_sha256":"91dfde512a311cbb9fcb25ef060e413301c38e72b02dc0cf2bbd75d8d0428e2f"},"source":{"id":"2604.18239","kind":"arxiv","version":4},"verdict":{"id":"a48068ab-ac79-4d83-9527-ed78ac14c6df","model_set":{"reader":"grok-4.3"},"created_at":"2026-05-10T04:56:54.838751Z","strongest_claim":"Using the DB, we propose reward calibration (RC), a plug-and-play method that adaptively rebalances the updates for chosen and rejected responses to satisfy the DB, without redesigning the base objective. Empirical results show that RC leads to more disentangled dynamics, with better downstream performance observed across several settings.","one_line_summary":"A unified incentive-score decomposition of preference optimization reveals the disentanglement band condition and reward calibration method that enables suppressing losers while preserving winners in LLM training.","pipeline_version":"pith-pipeline@v0.9.0","weakest_assumption":"The decomposition assumes that different objectives share the same local update directions and differ only in scalar weights, which must hold for the disentanglement band analysis to apply across methods.","pith_extraction_headline":"Preference optimization can suppress rejected responses without harming chosen ones by satisfying the disentanglement band condition."},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2604.18239/integrity.json","findings":[],"available":true,"detectors_run":[{"name":"doi_compliance","ran_at":"2026-05-20T04:16:28.744964Z","status":"completed","version":"1.0.0","findings_count":0}],"snapshot_sha256":"6b9b9f855fe2f5d09fefb5159641029a7f6a94637fddf1b255f9825907810419"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":"a48068ab-ac79-4d83-9527-ed78ac14c6df"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-24T01:24:11Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"KTsvhAYjLRoEvoKKAt9d3tP3yOe4EAf1H1nztaeFOjQ85vRPueNDnMsNVNH6wq6HzYiYBCMbENbzjmUY7WuHBQ==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-12T21:04:33.203617Z"},"content_sha256":"ba5b65beedd07fc1e71215fa03aaf0d69a00b8149f392ebcfea47d286bc2311d","schema_version":"1.0","event_id":"sha256:ba5b65beedd07fc1e71215fa03aaf0d69a00b8149f392ebcfea47d286bc2311d"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/7Q7GUWLHTTC46OZIG2RUPN3ERS/bundle.json","state_url":"https://pith.science/pith/7Q7GUWLHTTC46OZIG2RUPN3ERS/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/7Q7GUWLHTTC46OZIG2RUPN3ERS/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-12T21:04:33Z","links":{"resolver":"https://pith.science/pith/7Q7GUWLHTTC46OZIG2RUPN3ERS","bundle":"https://pith.science/pith/7Q7GUWLHTTC46OZIG2RUPN3ERS/bundle.json","state":"https://pith.science/pith/7Q7GUWLHTTC46OZIG2RUPN3ERS/state.json","well_known_bundle":"https://pith.science/.well-known/pith/7Q7GUWLHTTC46OZIG2RUPN3ERS/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2026:7Q7GUWLHTTC46OZIG2RUPN3ERS","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"bb641315dd1ec6ab5e83a45584826fbc89fded5f76ada51774d65df556bde155","cross_cats_sorted":["cs.AI"],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2026-04-20T13:23:27Z","title_canon_sha256":"ed324f6d54726228edb39bb3947162d0bdd4409f98aff613834639a5f762f8b9"},"schema_version":"1.0","source":{"id":"2604.18239","kind":"arxiv","version":4}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2604.18239","created_at":"2026-07-24T01:24:11Z"},{"alias_kind":"arxiv_version","alias_value":"2604.18239v4","created_at":"2026-07-24T01:24:11Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2604.18239","created_at":"2026-07-24T01:24:11Z"},{"alias_kind":"pith_short_12","alias_value":"7Q7GUWLHTTC4","created_at":"2026-07-24T01:24:11Z"},{"alias_kind":"pith_short_16","alias_value":"7Q7GUWLHTTC46OZI","created_at":"2026-07-24T01:24:11Z"},{"alias_kind":"pith_short_8","alias_value":"7Q7GUWLH","created_at":"2026-07-24T01:24:11Z"}],"graph_snapshots":[{"event_id":"sha256:ba5b65beedd07fc1e71215fa03aaf0d69a00b8149f392ebcfea47d286bc2311d","target":"graph","created_at":"2026-07-24T01:24:11Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":4,"items":[{"attestation":"unclaimed","claim_id":"C1","kind":"strongest_claim","source":"verdict.strongest_claim","status":"machine_extracted","text":"Using the DB, we propose reward calibration (RC), a plug-and-play method that adaptively rebalances the updates for chosen and rejected responses to satisfy the DB, without redesigning the base objective. Empirical results show that RC leads to more disentangled dynamics, with better downstream performance observed across several settings."},{"attestation":"unclaimed","claim_id":"C2","kind":"weakest_assumption","source":"verdict.weakest_assumption","status":"machine_extracted","text":"The decomposition assumes that different objectives share the same local update directions and differ only in scalar weights, which must hold for the disentanglement band analysis to apply across methods."},{"attestation":"unclaimed","claim_id":"C3","kind":"one_line_summary","source":"verdict.one_line_summary","status":"machine_extracted","text":"A unified incentive-score decomposition of preference optimization reveals the disentanglement band condition and reward calibration method that enables suppressing losers while preserving winners in LLM training."},{"attestation":"unclaimed","claim_id":"C4","kind":"headline","source":"verdict.pith_extraction.headline","status":"machine_extracted","text":"Preference optimization can suppress rejected responses without harming chosen ones by satisfying the disentanglement band condition."}],"snapshot_sha256":"91dfde512a311cbb9fcb25ef060e413301c38e72b02dc0cf2bbd75d8d0428e2f"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[{"findings_count":0,"name":"doi_compliance","ran_at":"2026-05-20T04:16:28.744964Z","status":"completed","version":"1.0.0"}],"endpoint":"/pith/2604.18239/integrity.json","findings":[],"snapshot_sha256":"6b9b9f855fe2f5d09fefb5159641029a7f6a94637fddf1b255f9825907810419","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Preference optimization is widely used to align large language models (LLMs) with human preferences. However, many margin-based methods also suppress the chosen response when they try to suppress the rejected one, and there is no general way to prevent this across different objectives.\n  We address this issue with a unified incentive-score decomposition of preference optimization, revealing that different objectives share the same local update directions and differ only in their scalar weights.\n  This decomposition provides a common framework for analyzing objectives that were previously studi","authors_text":"Delu Zeng, John Paisley, Junmei Yang, Min Chen, Qibin Zhao, Wei Chen, Yubing Wu, Zhou Wang","cross_cats":["cs.AI"],"headline":"Preference optimization can suppress rejected responses without harming chosen ones by satisfying the disentanglement band condition.","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2026-04-20T13:23:27Z","title":"Towards Disentangled Preference Optimization Dynamics: Suppress the Loser, Preserve the Winner"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2604.18239","kind":"arxiv","version":4},"verdict":{"created_at":"2026-05-10T04:56:54.838751Z","id":"a48068ab-ac79-4d83-9527-ed78ac14c6df","model_set":{"reader":"grok-4.3"},"one_line_summary":"A unified incentive-score decomposition of preference optimization reveals the disentanglement band condition and reward calibration method that enables suppressing losers while preserving winners in LLM training.","pipeline_version":"pith-pipeline@v0.9.0","pith_extraction_headline":"Preference optimization can suppress rejected responses without harming chosen ones by satisfying the disentanglement band condition.","strongest_claim":"Using the DB, we propose reward calibration (RC), a plug-and-play method that adaptively rebalances the updates for chosen and rejected responses to satisfy the DB, without redesigning the base objective. Empirical results show that RC leads to more disentangled dynamics, with better downstream performance observed across several settings.","weakest_assumption":"The decomposition assumes that different objectives share the same local update directions and differ only in scalar weights, which must hold for the disentanglement band analysis to apply across methods."}},"verdict_id":"a48068ab-ac79-4d83-9527-ed78ac14c6df"}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:9037f5855c7b707cd24c9874cc1959835ad5567ec5790e2c0dbeed9fe643115e","target":"record","created_at":"2026-07-24T01:24:11Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"bb641315dd1ec6ab5e83a45584826fbc89fded5f76ada51774d65df556bde155","cross_cats_sorted":["cs.AI"],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2026-04-20T13:23:27Z","title_canon_sha256":"ed324f6d54726228edb39bb3947162d0bdd4409f98aff613834639a5f762f8b9"},"schema_version":"1.0","source":{"id":"2604.18239","kind":"arxiv","version":4}},"canonical_sha256":"fc3e6a59679cc5cf3b2836a347b7648c809cb6022e016e42d3c89538445328b8","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"fc3e6a59679cc5cf3b2836a347b7648c809cb6022e016e42d3c89538445328b8","first_computed_at":"2026-07-24T01:24:11.398310Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-24T01:24:11.398310Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"5GINvnQmkvNxU6/ZiXrX3KmV4tl1ZN+RFDgKx2AiOFFSgrNH8J7zy/U6L9m9tgDFmmI5JVJERjFTGBskobf+Dg==","signature_status":"signed_v1","signed_at":"2026-07-24T01:24:11.399258Z","signed_message":"canonical_sha256_bytes"},"source_id":"2604.18239","source_kind":"arxiv","source_version":4}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:9037f5855c7b707cd24c9874cc1959835ad5567ec5790e2c0dbeed9fe643115e","sha256:ba5b65beedd07fc1e71215fa03aaf0d69a00b8149f392ebcfea47d286bc2311d"],"state_sha256":"eaeb4ab923df9c65bf5dd9773e1c7a546e609f4782042453de7c57c72fb644ec"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"WoHOK2JKvJaz92dqtXcX+4W4rpNteeN2rFP0YSsW8jGsXm/8FEYqukBEjD/qSuRDMYkb4S+g4HcsfALNJzcrBQ==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-12T21:04:33.208839Z","bundle_sha256":"d84ae601c75becd488eec8ac7eff34e77ec4ddae317a60f985213d1435ad66f9"}}