{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:2X7HETXTTYYDYDLC4HNLLOTCLI","short_pith_number":"pith:2X7HETXT","schema_version":"1.0","canonical_sha256":"d5fe724ef39e303c0d62e1dab5ba625a1c171acc44fc22a194731af0ebad053b","source":{"kind":"arxiv","id":"2410.05255","version":2},"attestation_state":"computed","paper":{"title":"Bridging SFT and DPO for Diffusion Model Alignment with Self-Sampling Preference Optimization","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CV","authors_text":"Christopher Brinton, Daoan Zhang, Dong-Jun Han, Guangchen Lan, Hongming Zhang, Jiebo Luo, Mingxiao Li, Pengcheng Chen, Wenlin Yao, Xiaoman Pan, Yu Dong","submitted_at":"2024-10-07T17:56:53Z","abstract_excerpt":"Existing post-training techniques are broadly categorized into supervised fine-tuning (SFT) and reinforcement learning (RL) methods; the former is stable during training but suffers from limited generalization, while the latter, despite its stronger generalization capability, relies on additional preference data or reward models and carries the risk of reward exploitation. In order to preserve the advantages of both SFT and RL -- namely, eliminating the need for paired data and reward models while retaining the training stability of SFT and the generalization ability of RL -- a new alignment m"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.05255","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2024-10-07T17:56:53Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"c6085ab4a763825d44234158a9c7b6bc0428c7b409c652d61bde4061ee2cb054","abstract_canon_sha256":"2038b41f05ce24fd9340813668519dad71433fbf32f93322a34d3deab5534525"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:29:51.184335Z","signature_b64":"rPXPfz0Jbqj8T0gZSQhz815ij3e9BrqlDx56iVS8Ol0UHXd3hjyNP/tp/stkJBzaTaBkdhYpZPGpGzli0WebAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"d5fe724ef39e303c0d62e1dab5ba625a1c171acc44fc22a194731af0ebad053b","last_reissued_at":"2026-07-05T11:29:51.183800Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:29:51.183800Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Bridging SFT and DPO for Diffusion Model Alignment with Self-Sampling Preference Optimization","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CV","authors_text":"Christopher Brinton, Daoan Zhang, Dong-Jun Han, Guangchen Lan, Hongming Zhang, Jiebo Luo, Mingxiao Li, Pengcheng Chen, Wenlin Yao, Xiaoman Pan, Yu Dong","submitted_at":"2024-10-07T17:56:53Z","abstract_excerpt":"Existing post-training techniques are broadly categorized into supervised fine-tuning (SFT) and reinforcement learning (RL) methods; the former is stable during training but suffers from limited generalization, while the latter, despite its stronger generalization capability, relies on additional preference data or reward models and carries the risk of reward exploitation. In order to preserve the advantages of both SFT and RL -- namely, eliminating the need for paired data and reward models while retaining the training stability of SFT and the generalization ability of RL -- a new alignment m"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.05255","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.05255/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.05255","created_at":"2026-07-05T11:29:51.183865+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.05255v2","created_at":"2026-07-05T11:29:51.183865+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.05255","created_at":"2026-07-05T11:29:51.183865+00:00"},{"alias_kind":"pith_short_12","alias_value":"2X7HETXTTYYD","created_at":"2026-07-05T11:29:51.183865+00:00"},{"alias_kind":"pith_short_16","alias_value":"2X7HETXTTYYDYDLC","created_at":"2026-07-05T11:29:51.183865+00:00"},{"alias_kind":"pith_short_8","alias_value":"2X7HETXT","created_at":"2026-07-05T11:29:51.183865+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.08378","citing_title":"Reinforcement Learning for Scalable and Trustworthy Intelligent Systems","ref_index":195,"is_internal_anchor":false},{"citing_arxiv_id":"2604.24953","citing_title":"ViPO: Visual Preference Optimization at Scale","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2604.15311","citing_title":"LeapAlign: Post-Training Flow Matching Models at Any Generation Step by Building Two-Step Trajectories","ref_index":60,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/2X7HETXTTYYDYDLC4HNLLOTCLI","json":"https://pith.science/pith/2X7HETXTTYYDYDLC4HNLLOTCLI.json","graph_json":"https://pith.science/api/pith-number/2X7HETXTTYYDYDLC4HNLLOTCLI/graph.json","events_json":"https://pith.science/api/pith-number/2X7HETXTTYYDYDLC4HNLLOTCLI/events.json","paper":"https://pith.science/paper/2X7HETXT"},"agent_actions":{"view_html":"https://pith.science/pith/2X7HETXTTYYDYDLC4HNLLOTCLI","download_json":"https://pith.science/pith/2X7HETXTTYYDYDLC4HNLLOTCLI.json","view_paper":"https://pith.science/paper/2X7HETXT","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.05255&json=true","fetch_graph":"https://pith.science/api/pith-number/2X7HETXTTYYDYDLC4HNLLOTCLI/graph.json","fetch_events":"https://pith.science/api/pith-number/2X7HETXTTYYDYDLC4HNLLOTCLI/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/2X7HETXTTYYDYDLC4HNLLOTCLI/action/timestamp_anchor","attest_storage":"https://pith.science/pith/2X7HETXTTYYDYDLC4HNLLOTCLI/action/storage_attestation","attest_author":"https://pith.science/pith/2X7HETXTTYYDYDLC4HNLLOTCLI/action/author_attestation","sign_citation":"https://pith.science/pith/2X7HETXTTYYDYDLC4HNLLOTCLI/action/citation_signature","submit_replication":"https://pith.science/pith/2X7HETXTTYYDYDLC4HNLLOTCLI/action/replication_record"}},"created_at":"2026-07-05T11:29:51.183865+00:00","updated_at":"2026-07-05T11:29:51.183865+00:00"}