{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:RRY34ZER6BVF3UOYRXW57NTUR6","short_pith_number":"pith:RRY34ZER","schema_version":"1.0","canonical_sha256":"8c71be6491f06a5dd1d88deddfb6748f9b33defb6bffd3ffaa2923354e61ca7e","source":{"kind":"arxiv","id":"2407.13734","version":1},"attestation_state":"computed","paper":{"title":"Understanding Reinforcement Learning-Based Fine-Tuning of Diffusion Models: A Tutorial and Review","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","q-bio.QM","stat.ML"],"primary_cat":"cs.LG","authors_text":"Masatoshi Uehara, Sergey Levine, Tommaso Biancalani, Yulai Zhao","submitted_at":"2024-07-18T17:35:32Z","abstract_excerpt":"This tutorial provides a comprehensive survey of methods for fine-tuning diffusion models to optimize downstream reward functions. While diffusion models are widely known to provide excellent generative modeling capability, practical applications in domains such as biology require generating samples that maximize some desired metric (e.g., translation efficiency in RNA, docking score in molecules, stability in protein). In these cases, the diffusion model can be optimized not only to generate realistic samples but also to explicitly maximize the measure of interest. Such methods are based on c"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2407.13734","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-07-18T17:35:32Z","cross_cats_sorted":["cs.AI","q-bio.QM","stat.ML"],"title_canon_sha256":"462271bd8cf495b4cf4ddf2a1c40de7efa10125cd09344de93db5cd6b6469857","abstract_canon_sha256":"15b8bfdb8dad70170b43f5393f0dc071ce2e4d6075e849e3de48325e5675410f"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:45:43.057978Z","signature_b64":"jqRthDBf0jH69Hhiortp6wwOXNeyfh32FjeipJAujmJEdsw7DIg0fLFAr+Oahr2rNP7czkT+UX1sYADJIze2BQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"8c71be6491f06a5dd1d88deddfb6748f9b33defb6bffd3ffaa2923354e61ca7e","last_reissued_at":"2026-07-05T08:45:43.057559Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:45:43.057559Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Understanding Reinforcement Learning-Based Fine-Tuning of Diffusion Models: A Tutorial and Review","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","q-bio.QM","stat.ML"],"primary_cat":"cs.LG","authors_text":"Masatoshi Uehara, Sergey Levine, Tommaso Biancalani, Yulai Zhao","submitted_at":"2024-07-18T17:35:32Z","abstract_excerpt":"This tutorial provides a comprehensive survey of methods for fine-tuning diffusion models to optimize downstream reward functions. While diffusion models are widely known to provide excellent generative modeling capability, practical applications in domains such as biology require generating samples that maximize some desired metric (e.g., translation efficiency in RNA, docking score in molecules, stability in protein). In these cases, the diffusion model can be optimized not only to generate realistic samples but also to explicitly maximize the measure of interest. Such methods are based on c"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2407.13734","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2407.13734/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2407.13734","created_at":"2026-07-05T08:45:43.057617+00:00"},{"alias_kind":"arxiv_version","alias_value":"2407.13734v1","created_at":"2026-07-05T08:45:43.057617+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2407.13734","created_at":"2026-07-05T08:45:43.057617+00:00"},{"alias_kind":"pith_short_12","alias_value":"RRY34ZER6BVF","created_at":"2026-07-05T08:45:43.057617+00:00"},{"alias_kind":"pith_short_16","alias_value":"RRY34ZER6BVF3UOY","created_at":"2026-07-05T08:45:43.057617+00:00"},{"alias_kind":"pith_short_8","alias_value":"RRY34ZER","created_at":"2026-07-05T08:45:43.057617+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":26,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2604.17415","citing_title":"Reward Score Matching: Unifying Reward-based Fine-tuning for Flow and Diffusion Models","ref_index":49,"is_internal_anchor":true},{"citing_arxiv_id":"2606.22652","citing_title":"A Markov Chain Approach to Preference Alignment","ref_index":50,"is_internal_anchor":false},{"citing_arxiv_id":"2606.21014","citing_title":"BayesFP: Posterior Estimation for Flow-Based Policies via Feynman-Kac Sampling","ref_index":38,"is_internal_anchor":false},{"citing_arxiv_id":"2606.13565","citing_title":"A2D2: Fine-Tuning Any-Length Discrete Diffusion for Adaptive Decoding","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2606.08802","citing_title":"Active Flow Expansion for Out-of-Distribution Discovery: from Theory to Molecules","ref_index":53,"is_internal_anchor":false},{"citing_arxiv_id":"2606.06303","citing_title":"Plug-and-Play Guidance for Discrete Diffusion Models via Gradient-Informed Logit Correction","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2606.06272","citing_title":"Your GFlowNet Secretly Learns an Optimal Transport Plan","ref_index":143,"is_internal_anchor":false},{"citing_arxiv_id":"2604.27147","citing_title":"How to Guide Your Flow: Few-Step Alignment via Flow Map Reward Guidance","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2606.30414","citing_title":"Diffusion Fine-tuning with Rewarded Moment Matching Distillation","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2606.28578","citing_title":"Surrogate-Gated Generation and Foundation-Model Embeddings for Bayesian Materials Design","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2605.26013","citing_title":"AdvantageFlow: Advantage-Weighted Least Squares for RL in Flow Models","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2605.26552","citing_title":"Aligning Few-Step Generative Models by Amortizing Sample-based Variational Inference","ref_index":93,"is_internal_anchor":false},{"citing_arxiv_id":"2605.30046","citing_title":"Masked Diffusion Modeling for Anomaly Detection","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2605.23346","citing_title":"Contrastive Distribution Matching for Amortized Sequential Monte Carlo in Discrete Diffusion","ref_index":77,"is_internal_anchor":false},{"citing_arxiv_id":"2604.27147","citing_title":"How to Guide Your Flow: Few-Step Alignment via Flow Map Reward Guidance","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2605.15855","citing_title":"Do Less, Achieve More: Do We Need Every-Step Optimization for RL Fine-tuning of Diffusion Models?","ref_index":51,"is_internal_anchor":false},{"citing_arxiv_id":"2510.06637","citing_title":"Control-Augmented Autoregressive Diffusion for Data Assimilation","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2602.02924","citing_title":"How Does the Lagrangian Guide Safe Reinforcement Learning through Diffusion Models?","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11347","citing_title":"Gradient-Free Noise Optimization for Reward Alignment in Generative Models","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11347","citing_title":"Gradient-Free Noise Optimization for Reward Alignment in Generative Models","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12112","citing_title":"When Policy Entropy Constraint Fails: Preserving Diversity in Flow-based RLHF via Perceptual Entropy","ref_index":57,"is_internal_anchor":false},{"citing_arxiv_id":"2604.27147","citing_title":"How to Guide Your Flow: Few-Step Alignment via Flow Map Reward Guidance","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06583","citing_title":"Improved techniques for fine-tuning flow models via adjoint matching: a deterministic control pipeline","ref_index":39,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05387","citing_title":"Conditional Diffusion Under Linear Constraints: Langevin Mixing and Information-Theoretic Guarantees","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2604.14379","citing_title":"Step-level Denoising-time Diffusion Alignment with Multiple Objectives","ref_index":31,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/RRY34ZER6BVF3UOYRXW57NTUR6","json":"https://pith.science/pith/RRY34ZER6BVF3UOYRXW57NTUR6.json","graph_json":"https://pith.science/api/pith-number/RRY34ZER6BVF3UOYRXW57NTUR6/graph.json","events_json":"https://pith.science/api/pith-number/RRY34ZER6BVF3UOYRXW57NTUR6/events.json","paper":"https://pith.science/paper/RRY34ZER"},"agent_actions":{"view_html":"https://pith.science/pith/RRY34ZER6BVF3UOYRXW57NTUR6","download_json":"https://pith.science/pith/RRY34ZER6BVF3UOYRXW57NTUR6.json","view_paper":"https://pith.science/paper/RRY34ZER","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2407.13734&json=true","fetch_graph":"https://pith.science/api/pith-number/RRY34ZER6BVF3UOYRXW57NTUR6/graph.json","fetch_events":"https://pith.science/api/pith-number/RRY34ZER6BVF3UOYRXW57NTUR6/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/RRY34ZER6BVF3UOYRXW57NTUR6/action/timestamp_anchor","attest_storage":"https://pith.science/pith/RRY34ZER6BVF3UOYRXW57NTUR6/action/storage_attestation","attest_author":"https://pith.science/pith/RRY34ZER6BVF3UOYRXW57NTUR6/action/author_attestation","sign_citation":"https://pith.science/pith/RRY34ZER6BVF3UOYRXW57NTUR6/action/citation_signature","submit_replication":"https://pith.science/pith/RRY34ZER6BVF3UOYRXW57NTUR6/action/replication_record"}},"created_at":"2026-07-05T08:45:43.057617+00:00","updated_at":"2026-07-05T08:45:43.057617+00:00"}