{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:464BD74NRSN4BFK4BIUTJB7TZF","short_pith_number":"pith:464BD74N","schema_version":"1.0","canonical_sha256":"e7b811ff8d8c9bc0955c0a293487f3c96a6c06132d9dadba0447ad8592d18e43","source":{"kind":"arxiv","id":"2402.08265","version":2},"attestation_state":"computed","paper":{"title":"A Dense Reward View on Aligning Text-to-Image Diffusion with Preference","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Mingyuan Zhou, Shentao Yang, Tianqi Chen","submitted_at":"2024-02-13T07:37:24Z","abstract_excerpt":"Aligning text-to-image diffusion model (T2I) with preference has been gaining increasing research attention. While prior works exist on directly optimizing T2I by preference data, these methods are developed under the bandit assumption of a latent reward on the entire diffusion reverse chain, while ignoring the sequential nature of the generation process. This may harm the efficacy and efficiency of preference alignment. In this paper, we take on a finer dense reward perspective and derive a tractable alignment objective that emphasizes the initial steps of the T2I reverse chain. In particular"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2402.08265","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2024-02-13T07:37:24Z","cross_cats_sorted":[],"title_canon_sha256":"a479828b152cc12ef044493107339eef4c534d46dcf6a25246966c4ae8aece5b","abstract_canon_sha256":"ec7eadf9ac1217c52648528ed268297413aa400beb5b546bda48d424eb830a00"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:18:05.379616Z","signature_b64":"VhuBnhLp7Mrj++N3LLU1p68ghE7MZ/6/zWMCrqtbM9J/jgXVS0ARqoT2a+IIR/vrF8Djjq4SEMS1FpRkjcZPBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"e7b811ff8d8c9bc0955c0a293487f3c96a6c06132d9dadba0447ad8592d18e43","last_reissued_at":"2026-07-05T08:18:05.379090Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:18:05.379090Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"A Dense Reward View on Aligning Text-to-Image Diffusion with Preference","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CV","authors_text":"Mingyuan Zhou, Shentao Yang, Tianqi Chen","submitted_at":"2024-02-13T07:37:24Z","abstract_excerpt":"Aligning text-to-image diffusion model (T2I) with preference has been gaining increasing research attention. While prior works exist on directly optimizing T2I by preference data, these methods are developed under the bandit assumption of a latent reward on the entire diffusion reverse chain, while ignoring the sequential nature of the generation process. This may harm the efficacy and efficiency of preference alignment. In this paper, we take on a finer dense reward perspective and derive a tractable alignment objective that emphasizes the initial steps of the T2I reverse chain. In particular"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2402.08265","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2402.08265/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2402.08265","created_at":"2026-07-05T08:18:05.379144+00:00"},{"alias_kind":"arxiv_version","alias_value":"2402.08265v2","created_at":"2026-07-05T08:18:05.379144+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2402.08265","created_at":"2026-07-05T08:18:05.379144+00:00"},{"alias_kind":"pith_short_12","alias_value":"464BD74NRSN4","created_at":"2026-07-05T08:18:05.379144+00:00"},{"alias_kind":"pith_short_16","alias_value":"464BD74NRSN4BFK4","created_at":"2026-07-05T08:18:05.379144+00:00"},{"alias_kind":"pith_short_8","alias_value":"464BD74N","created_at":"2026-07-05T08:18:05.379144+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":7,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2604.18518","citing_title":"UDM-GRPO: Stable and Efficient Group Relative Policy Optimization for Uniform Discrete Diffusion Models","ref_index":49,"is_internal_anchor":true},{"citing_arxiv_id":"2605.27736","citing_title":"Explicit Critic Guidance for Aligning Diffusion Models","ref_index":86,"is_internal_anchor":false},{"citing_arxiv_id":"2605.21573","citing_title":"Lens: Rethinking Training Efficiency for Foundational Text-to-Image Models","ref_index":50,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09433","citing_title":"Offline Preference Optimization for Rectified Flow with Noise-Tracked Pairs","ref_index":46,"is_internal_anchor":false},{"citing_arxiv_id":"2605.04653","citing_title":"Threshold-Guided Optimization for Visual Generative Models","ref_index":41,"is_internal_anchor":false},{"citing_arxiv_id":"2604.15311","citing_title":"LeapAlign: Post-Training Flow Matching Models at Any Generation Step by Building Two-Step Trajectories","ref_index":57,"is_internal_anchor":false},{"citing_arxiv_id":"2604.18518","citing_title":"UDM-GRPO: Stable and Efficient Group Relative Policy Optimization for Uniform Discrete Diffusion Models","ref_index":24,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/464BD74NRSN4BFK4BIUTJB7TZF","json":"https://pith.science/pith/464BD74NRSN4BFK4BIUTJB7TZF.json","graph_json":"https://pith.science/api/pith-number/464BD74NRSN4BFK4BIUTJB7TZF/graph.json","events_json":"https://pith.science/api/pith-number/464BD74NRSN4BFK4BIUTJB7TZF/events.json","paper":"https://pith.science/paper/464BD74N"},"agent_actions":{"view_html":"https://pith.science/pith/464BD74NRSN4BFK4BIUTJB7TZF","download_json":"https://pith.science/pith/464BD74NRSN4BFK4BIUTJB7TZF.json","view_paper":"https://pith.science/paper/464BD74N","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2402.08265&json=true","fetch_graph":"https://pith.science/api/pith-number/464BD74NRSN4BFK4BIUTJB7TZF/graph.json","fetch_events":"https://pith.science/api/pith-number/464BD74NRSN4BFK4BIUTJB7TZF/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/464BD74NRSN4BFK4BIUTJB7TZF/action/timestamp_anchor","attest_storage":"https://pith.science/pith/464BD74NRSN4BFK4BIUTJB7TZF/action/storage_attestation","attest_author":"https://pith.science/pith/464BD74NRSN4BFK4BIUTJB7TZF/action/author_attestation","sign_citation":"https://pith.science/pith/464BD74NRSN4BFK4BIUTJB7TZF/action/citation_signature","submit_replication":"https://pith.science/pith/464BD74NRSN4BFK4BIUTJB7TZF/action/replication_record"}},"created_at":"2026-07-05T08:18:05.379144+00:00","updated_at":"2026-07-05T08:18:05.379144+00:00"}