{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:YSHUYZCMQKGYXMUHLDYMOLY2C4","short_pith_number":"pith:YSHUYZCM","schema_version":"1.0","canonical_sha256":"c48f4c644c828d8bb28758f0c72f1a172b9412a0739989fdd53f5eb4de9e88e9","source":{"kind":"arxiv","id":"2310.03739","version":5},"attestation_state":"computed","paper":{"title":"Aligning Text-to-Image Diffusion Models with Reward Backpropagation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG","cs.RO"],"primary_cat":"cs.CV","authors_text":"Anirudh Goyal, Deepak Pathak, Katerina Fragkiadaki, Mihir Prabhudesai","submitted_at":"2023-10-05T17:59:18Z","abstract_excerpt":"Text-to-image diffusion models have recently emerged at the forefront of image generation, powered by very large-scale unsupervised or weakly supervised text-to-image training datasets. Due to their unsupervised training, controlling their behavior in downstream tasks, such as maximizing human-perceived image quality, image-text alignment, or ethical image generation, is difficult. Recent works finetune diffusion models to downstream reward functions using vanilla reinforcement learning, notorious for the high variance of the gradient estimators. In this paper, we propose AlignProp, a method t"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2310.03739","kind":"arxiv","version":5},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CV","submitted_at":"2023-10-05T17:59:18Z","cross_cats_sorted":["cs.AI","cs.LG","cs.RO"],"title_canon_sha256":"642c2a71ec546ebe9d567287e9affdd7d3bf71008dd3a28e7af581c77e7b04ab","abstract_canon_sha256":"8800706c54980c16eac53f069894a7141e8f6365b1b8532880635c681735a422"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:32:03.745919Z","signature_b64":"mO2G9/XmzLGbfXEkGZtex++zY3cbNfLcR0yeGkqMrmyeiuF9+WhHfD43cDl0atmq2WvNyNZur/QvhzUT/O1+CQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"c48f4c644c828d8bb28758f0c72f1a172b9412a0739989fdd53f5eb4de9e88e9","last_reissued_at":"2026-07-05T09:32:03.745344Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:32:03.745344Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Aligning Text-to-Image Diffusion Models with Reward Backpropagation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG","cs.RO"],"primary_cat":"cs.CV","authors_text":"Anirudh Goyal, Deepak Pathak, Katerina Fragkiadaki, Mihir Prabhudesai","submitted_at":"2023-10-05T17:59:18Z","abstract_excerpt":"Text-to-image diffusion models have recently emerged at the forefront of image generation, powered by very large-scale unsupervised or weakly supervised text-to-image training datasets. Due to their unsupervised training, controlling their behavior in downstream tasks, such as maximizing human-perceived image quality, image-text alignment, or ethical image generation, is difficult. Recent works finetune diffusion models to downstream reward functions using vanilla reinforcement learning, notorious for the high variance of the gradient estimators. In this paper, we propose AlignProp, a method t"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2310.03739","kind":"arxiv","version":5},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2310.03739/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2310.03739","created_at":"2026-07-05T09:32:03.745412+00:00"},{"alias_kind":"arxiv_version","alias_value":"2310.03739v5","created_at":"2026-07-05T09:32:03.745412+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2310.03739","created_at":"2026-07-05T09:32:03.745412+00:00"},{"alias_kind":"pith_short_12","alias_value":"YSHUYZCMQKGY","created_at":"2026-07-05T09:32:03.745412+00:00"},{"alias_kind":"pith_short_16","alias_value":"YSHUYZCMQKGYXMUH","created_at":"2026-07-05T09:32:03.745412+00:00"},{"alias_kind":"pith_short_8","alias_value":"YSHUYZCM","created_at":"2026-07-05T09:32:03.745412+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":22,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.24888","citing_title":"DiffusionBench: On Holistic Evaluation of Diffusion Transformers","ref_index":110,"is_internal_anchor":false},{"citing_arxiv_id":"2606.27377","citing_title":"DanceOPD: On-Policy Generative Field Distillation","ref_index":76,"is_internal_anchor":false},{"citing_arxiv_id":"2606.23626","citing_title":"DiT-Reward: Generative Representations for Text-to-Image Reward Modeling","ref_index":109,"is_internal_anchor":false},{"citing_arxiv_id":"2606.19162","citing_title":"The Reward Was in Your Data All Along: Correcting Flow Matching with Discriminator-Guided RL","ref_index":47,"is_internal_anchor":false},{"citing_arxiv_id":"2606.17979","citing_title":"STAR: SpatioTemporal Adaptive Reward Allocation for Text-to-Image RL Post-Training","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2606.06303","citing_title":"Plug-and-Play Guidance for Discrete Diffusion Models via Gradient-Informed Logit Correction","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2606.30376","citing_title":"FlowAWR: Online Adaptive Flow Reinforcement via Advantage-Weighted Rectification","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2605.26013","citing_title":"AdvantageFlow: Advantage-Weighted Least Squares for RL in Flow Models","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2605.30991","citing_title":"Parallel Tempering Initial Sampling in Inference-Time Reward Alignment","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2602.11146","citing_title":"Beyond VLM-Based Rewards: Diffusion-Native Latent Reward Modeling","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2505.17352","citing_title":"Alignment and Safety of Diffusion Models via Reinforcement Learning and Reward Modeling: A Survey","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2605.21195","citing_title":"RankE: End-to-End Post-Training for Discrete Text-to-Image Generation with Decoder Co-Evolution","ref_index":46,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10759","citing_title":"Reinforce Adjoint Matching: Scaling RL Post-Training of Diffusion and Flow-Matching Models","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11480","citing_title":"Efficient Adjoint Matching for Fine-tuning Diffusion Models","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2512.01236","citing_title":"PSR: Scaling Multi-Subject Personalized Image Generation with Pairwise Subject-Consistency Rewards","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2604.08580","citing_title":"Adjoint Matching through the Lens of the Stochastic Maximum Principle in Optimal Control","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2501.13918","citing_title":"Improving Video Generation with Human Feedback","ref_index":58,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11480","citing_title":"Efficient Adjoint Matching for Fine-tuning Diffusion Models","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2604.28185","citing_title":"Visual Generation in the New Era: An Evolution from Atomic Mapping to Agentic World Modeling","ref_index":61,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10759","citing_title":"Reinforce Adjoint Matching: Scaling RL Post-Training of Diffusion and Flow-Matching Models","ref_index":52,"is_internal_anchor":false},{"citing_arxiv_id":"2505.05470","citing_title":"Flow-GRPO: Training Flow Matching Models via Online RL","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2604.07427","citing_title":"Personalizing Text-to-Image Generation to Individual Taste","ref_index":42,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/YSHUYZCMQKGYXMUHLDYMOLY2C4","json":"https://pith.science/pith/YSHUYZCMQKGYXMUHLDYMOLY2C4.json","graph_json":"https://pith.science/api/pith-number/YSHUYZCMQKGYXMUHLDYMOLY2C4/graph.json","events_json":"https://pith.science/api/pith-number/YSHUYZCMQKGYXMUHLDYMOLY2C4/events.json","paper":"https://pith.science/paper/YSHUYZCM"},"agent_actions":{"view_html":"https://pith.science/pith/YSHUYZCMQKGYXMUHLDYMOLY2C4","download_json":"https://pith.science/pith/YSHUYZCMQKGYXMUHLDYMOLY2C4.json","view_paper":"https://pith.science/paper/YSHUYZCM","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2310.03739&json=true","fetch_graph":"https://pith.science/api/pith-number/YSHUYZCMQKGYXMUHLDYMOLY2C4/graph.json","fetch_events":"https://pith.science/api/pith-number/YSHUYZCMQKGYXMUHLDYMOLY2C4/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/YSHUYZCMQKGYXMUHLDYMOLY2C4/action/timestamp_anchor","attest_storage":"https://pith.science/pith/YSHUYZCMQKGYXMUHLDYMOLY2C4/action/storage_attestation","attest_author":"https://pith.science/pith/YSHUYZCMQKGYXMUHLDYMOLY2C4/action/author_attestation","sign_citation":"https://pith.science/pith/YSHUYZCMQKGYXMUHLDYMOLY2C4/action/citation_signature","submit_replication":"https://pith.science/pith/YSHUYZCMQKGYXMUHLDYMOLY2C4/action/replication_record"}},"created_at":"2026-07-05T09:32:03.745412+00:00","updated_at":"2026-07-05T09:32:03.745412+00:00"}