{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:6MR7RJEMB6A25BSDBNHX3OINGB","short_pith_number":"pith:6MR7RJEM","schema_version":"1.0","canonical_sha256":"f323f8a48c0f81ae86430b4f7db90d30708cd794165c5c77679f707263f5a982","source":{"kind":"arxiv","id":"2505.11070","version":1},"attestation_state":"computed","paper":{"title":"Towards Self-Improvement of Diffusion Models via Group Preference Optimization","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Boyuan Liu, Chao Feng, Jiangchuan Wei, Jiao Ran, Mingyu Guo, Renjie Chen, Wenfeng Lin, Yichen Zhang","submitted_at":"2025-05-16T10:04:57Z","abstract_excerpt":"Aligning text-to-image (T2I) diffusion models with Direct Preference Optimization (DPO) has shown notable improvements in generation quality. However, applying DPO to T2I faces two challenges: the sensitivity of DPO to preference pairs and the labor-intensive process of collecting and annotating high-quality data. In this work, we demonstrate that preference pairs with marginal differences can degrade DPO performance. Since DPO relies exclusively on relative ranking while disregarding the absolute difference of pairs, it may misclassify losing samples as wins, or vice versa. We empirically sho"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.11070","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2025-05-16T10:04:57Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"63eb3291ca30b5d4281472fd6fc8efbcd85a1d1f9b045291004842ecc2ac362d","abstract_canon_sha256":"eb168c0040c84bf536c3d9b37364523bab329f7618b24c3825fbca0313779fa8"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:04:10.258933Z","signature_b64":"a2TeOh+fE9IRQDPPFBKzqrjKWqgD/pXNEF/1OetFqjRJ45V2V8Q7zREBZE5DS7REu/9ZasA1zKvNjPNIBY4RAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f323f8a48c0f81ae86430b4f7db90d30708cd794165c5c77679f707263f5a982","last_reissued_at":"2026-07-05T11:04:10.258450Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:04:10.258450Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Towards Self-Improvement of Diffusion Models via Group Preference Optimization","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CV","authors_text":"Boyuan Liu, Chao Feng, Jiangchuan Wei, Jiao Ran, Mingyu Guo, Renjie Chen, Wenfeng Lin, Yichen Zhang","submitted_at":"2025-05-16T10:04:57Z","abstract_excerpt":"Aligning text-to-image (T2I) diffusion models with Direct Preference Optimization (DPO) has shown notable improvements in generation quality. However, applying DPO to T2I faces two challenges: the sensitivity of DPO to preference pairs and the labor-intensive process of collecting and annotating high-quality data. In this work, we demonstrate that preference pairs with marginal differences can degrade DPO performance. Since DPO relies exclusively on relative ranking while disregarding the absolute difference of pairs, it may misclassify losing samples as wins, or vice versa. We empirically sho"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.11070","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.11070/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.11070","created_at":"2026-07-05T11:04:10.258513+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.11070v1","created_at":"2026-07-05T11:04:10.258513+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.11070","created_at":"2026-07-05T11:04:10.258513+00:00"},{"alias_kind":"pith_short_12","alias_value":"6MR7RJEMB6A2","created_at":"2026-07-05T11:04:10.258513+00:00"},{"alias_kind":"pith_short_16","alias_value":"6MR7RJEMB6A25BSD","created_at":"2026-07-05T11:04:10.258513+00:00"},{"alias_kind":"pith_short_8","alias_value":"6MR7RJEM","created_at":"2026-07-05T11:04:10.258513+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2603.22844","citing_title":"PhySe-RPO: Physics and Semantics Guided Relative Policy Optimization for Diffusion-Based Surgical Smoke Removal","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2604.06916","citing_title":"FP4 Explore, BF16 Train: Diffusion Reinforcement Learning via Efficient Rollout Scaling","ref_index":46,"is_internal_anchor":false},{"citing_arxiv_id":"2604.15311","citing_title":"LeapAlign: Post-Training Flow Matching Models at Any Generation Step by Building Two-Step Trajectories","ref_index":2,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/6MR7RJEMB6A25BSDBNHX3OINGB","json":"https://pith.science/pith/6MR7RJEMB6A25BSDBNHX3OINGB.json","graph_json":"https://pith.science/api/pith-number/6MR7RJEMB6A25BSDBNHX3OINGB/graph.json","events_json":"https://pith.science/api/pith-number/6MR7RJEMB6A25BSDBNHX3OINGB/events.json","paper":"https://pith.science/paper/6MR7RJEM"},"agent_actions":{"view_html":"https://pith.science/pith/6MR7RJEMB6A25BSDBNHX3OINGB","download_json":"https://pith.science/pith/6MR7RJEMB6A25BSDBNHX3OINGB.json","view_paper":"https://pith.science/paper/6MR7RJEM","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.11070&json=true","fetch_graph":"https://pith.science/api/pith-number/6MR7RJEMB6A25BSDBNHX3OINGB/graph.json","fetch_events":"https://pith.science/api/pith-number/6MR7RJEMB6A25BSDBNHX3OINGB/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/6MR7RJEMB6A25BSDBNHX3OINGB/action/timestamp_anchor","attest_storage":"https://pith.science/pith/6MR7RJEMB6A25BSDBNHX3OINGB/action/storage_attestation","attest_author":"https://pith.science/pith/6MR7RJEMB6A25BSDBNHX3OINGB/action/author_attestation","sign_citation":"https://pith.science/pith/6MR7RJEMB6A25BSDBNHX3OINGB/action/citation_signature","submit_replication":"https://pith.science/pith/6MR7RJEMB6A25BSDBNHX3OINGB/action/replication_record"}},"created_at":"2026-07-05T11:04:10.258513+00:00","updated_at":"2026-07-05T11:04:10.258513+00:00"}