{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:QZNDROGDYGBSTQN6NMFPEMAQAU","short_pith_number":"pith:QZNDROGD","schema_version":"1.0","canonical_sha256":"865a38b8c3c18329c1be6b0af23010053bbcf4365d0b3969a1700a9d44c73db3","source":{"kind":"arxiv","id":"2311.12908","version":1},"attestation_state":"computed","paper":{"title":"Diffusion Model Alignment Using Direct Preference Optimization","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.GR","cs.LG"],"primary_cat":"cs.CV","authors_text":"Aaron Lou, Bram Wallace, Caiming Xiong, Linqi Zhou, Meihua Dang, Nikhil Naik, Rafael Rafailov, Senthil Purushwalkam, Shafiq Joty, Stefano Ermon","submitted_at":"2023-11-21T15:24:05Z","abstract_excerpt":"Large language models (LLMs) are fine-tuned using human comparison data with Reinforcement Learning from Human Feedback (RLHF) methods to make them better aligned with users' preferences. In contrast to LLMs, human preference learning has not been widely explored in text-to-image diffusion models; the best existing approach is to fine-tune a pretrained model using carefully curated high quality images and captions to improve visual appeal and text alignment. We propose Diffusion-DPO, a method to align diffusion models to human preferences by directly optimizing on human comparison data. Diffus"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2311.12908","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2023-11-21T15:24:05Z","cross_cats_sorted":["cs.AI","cs.GR","cs.LG"],"title_canon_sha256":"0c48f642bd5cc2c59586651ce93c7e2aab0bfe119c32665bbf87a077ec140ce5","abstract_canon_sha256":"75e56cfefa7d9e84b624200cb77effc4b60bb4c3e3c532fff5bc9420194523f1"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:15:20.424730Z","signature_b64":"nDDiYCrhMo0agPGH5FfyGJuir2nSWp1UxWhwDKLBt2KmLgFIi4VYNwTFrVRRSQokIGXfiorW2vA6a6h2mI5yCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"865a38b8c3c18329c1be6b0af23010053bbcf4365d0b3969a1700a9d44c73db3","last_reissued_at":"2026-07-05T07:15:20.424204Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:15:20.424204Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Diffusion Model Alignment Using Direct Preference Optimization","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.GR","cs.LG"],"primary_cat":"cs.CV","authors_text":"Aaron Lou, Bram Wallace, Caiming Xiong, Linqi Zhou, Meihua Dang, Nikhil Naik, Rafael Rafailov, Senthil Purushwalkam, Shafiq Joty, Stefano Ermon","submitted_at":"2023-11-21T15:24:05Z","abstract_excerpt":"Large language models (LLMs) are fine-tuned using human comparison data with Reinforcement Learning from Human Feedback (RLHF) methods to make them better aligned with users' preferences. In contrast to LLMs, human preference learning has not been widely explored in text-to-image diffusion models; the best existing approach is to fine-tune a pretrained model using carefully curated high quality images and captions to improve visual appeal and text alignment. We propose Diffusion-DPO, a method to align diffusion models to human preferences by directly optimizing on human comparison data. Diffus"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2311.12908","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2311.12908/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2311.12908","created_at":"2026-07-05T07:15:20.424263+00:00"},{"alias_kind":"arxiv_version","alias_value":"2311.12908v1","created_at":"2026-07-05T07:15:20.424263+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2311.12908","created_at":"2026-07-05T07:15:20.424263+00:00"},{"alias_kind":"pith_short_12","alias_value":"QZNDROGDYGBS","created_at":"2026-07-05T07:15:20.424263+00:00"},{"alias_kind":"pith_short_16","alias_value":"QZNDROGDYGBSTQN6","created_at":"2026-07-05T07:15:20.424263+00:00"},{"alias_kind":"pith_short_8","alias_value":"QZNDROGD","created_at":"2026-07-05T07:15:20.424263+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":18,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.07693","citing_title":"Selective Timestep Weighting and Advantage-Based Replay for Sample-Efficient Diffusion RLHF","ref_index":44,"is_internal_anchor":true},{"citing_arxiv_id":"2606.22394","citing_title":"Curvature-Adaptive Consistency Flow Matching: Autonomous Trajectory Optimization via Reinforcement Learning","ref_index":57,"is_internal_anchor":false},{"citing_arxiv_id":"2606.20364","citing_title":"Judging to Improve: A De-biased VLM-as-3D-Judge Protocol for Single-Image 3D Generation","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2606.12434","citing_title":"Pluralistic-Alignment Urbanism: Operationalizing a Right to AI for Inclusive Public Space","ref_index":88,"is_internal_anchor":false},{"citing_arxiv_id":"2606.29150","citing_title":"Flow Reasoning Models: Scaling Reasoning Through Iterative Self-Refinement","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2510.03508","citing_title":"D2 Actor Critic: Diffusion Actor Meets Distributional Critic","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2605.23275","citing_title":"Diffusion Domain Expansion: Learning to Coordinate Pre-trained Diffusion Models","ref_index":54,"is_internal_anchor":false},{"citing_arxiv_id":"2505.17352","citing_title":"Alignment and Safety of Diffusion Models via Reinforcement Learning and Reward Modeling: A Survey","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2603.00607","citing_title":"IdGlow: Dynamic Identity Modulation for Multi-Subject Generation","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2406.03520","citing_title":"VideoPhy: Evaluating Physical Commonsense for Video Generation","ref_index":101,"is_internal_anchor":false},{"citing_arxiv_id":"2506.22832","citing_title":"Listener-Rewarded Thinking in VLMs for Image Preferences","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2509.11487","citing_title":"Collective Recourse for Generative Urban Visualizations","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2409.00588","citing_title":"Diffusion Policy Policy Optimization","ref_index":97,"is_internal_anchor":false},{"citing_arxiv_id":"2406.02430","citing_title":"Seed-TTS: A Family of High-Quality Versatile Speech Generation Models","ref_index":38,"is_internal_anchor":false},{"citing_arxiv_id":"2604.06779","citing_title":"VASR: Variance-Aware Systematic Resampling for Reward-Guided Diffusion","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2403.03206","citing_title":"Scaling Rectified Flow Transformers for High-Resolution Image Synthesis","ref_index":191,"is_internal_anchor":false},{"citing_arxiv_id":"2604.23536","citing_title":"$Z^2$-Sampling: Zero-Cost Zigzag Trajectories for Semantic Alignment in Diffusion Models","ref_index":44,"is_internal_anchor":false},{"citing_arxiv_id":"2604.06779","citing_title":"VASR: Variance-Aware Systematic Resampling for Reward-Guided Diffusion","ref_index":12,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/QZNDROGDYGBSTQN6NMFPEMAQAU","json":"https://pith.science/pith/QZNDROGDYGBSTQN6NMFPEMAQAU.json","graph_json":"https://pith.science/api/pith-number/QZNDROGDYGBSTQN6NMFPEMAQAU/graph.json","events_json":"https://pith.science/api/pith-number/QZNDROGDYGBSTQN6NMFPEMAQAU/events.json","paper":"https://pith.science/paper/QZNDROGD"},"agent_actions":{"view_html":"https://pith.science/pith/QZNDROGDYGBSTQN6NMFPEMAQAU","download_json":"https://pith.science/pith/QZNDROGDYGBSTQN6NMFPEMAQAU.json","view_paper":"https://pith.science/paper/QZNDROGD","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2311.12908&json=true","fetch_graph":"https://pith.science/api/pith-number/QZNDROGDYGBSTQN6NMFPEMAQAU/graph.json","fetch_events":"https://pith.science/api/pith-number/QZNDROGDYGBSTQN6NMFPEMAQAU/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/QZNDROGDYGBSTQN6NMFPEMAQAU/action/timestamp_anchor","attest_storage":"https://pith.science/pith/QZNDROGDYGBSTQN6NMFPEMAQAU/action/storage_attestation","attest_author":"https://pith.science/pith/QZNDROGDYGBSTQN6NMFPEMAQAU/action/author_attestation","sign_citation":"https://pith.science/pith/QZNDROGDYGBSTQN6NMFPEMAQAU/action/citation_signature","submit_replication":"https://pith.science/pith/QZNDROGDYGBSTQN6NMFPEMAQAU/action/replication_record"}},"created_at":"2026-07-05T07:15:20.424263+00:00","updated_at":"2026-07-05T07:15:20.424263+00:00"}