{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:E7YY3HPCR2VK3GAEQH7YCZWTMG","short_pith_number":"pith:E7YY3HPC","schema_version":"1.0","canonical_sha256":"27f18d9de28eaaad980481ff8166d361906e8dfbaa704a5c1a7cae289da77f59","source":{"kind":"arxiv","id":"2505.23458","version":1},"attestation_state":"computed","paper":{"title":"Diffusion Guidance Is a Controllable Policy Improvement Operator","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Kevin Frans, Pieter Abbeel, Seohong Park, Sergey Levine","submitted_at":"2025-05-29T14:06:50Z","abstract_excerpt":"At the core of reinforcement learning is the idea of learning beyond the performance in the data. However, scaling such systems has proven notoriously tricky. In contrast, techniques from generative modeling have proven remarkably scalable and are simple to train. In this work, we combine these strengths, by deriving a direct relation between policy improvement and guidance of diffusion models. The resulting framework, CFGRL, is trained with the simplicity of supervised learning, yet can further improve on the policies in the data. On offline RL tasks, we observe a reliable trend -- increased "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.23458","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2025-05-29T14:06:50Z","cross_cats_sorted":[],"title_canon_sha256":"5807fa1242d0f63dd464efb469d36b3d1af3b28642c0c3cf293d9d48ce5aa1c4","abstract_canon_sha256":"94250d12507002b66a24a31f6c9b7f30e1d3584a95f039310cf74efdbd5482e2"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:12:02.134104Z","signature_b64":"nCr1THhBbzT08jEjasV9pWxYMixx1mBMX1NLXYKGiIe9wJHao0dMG2TutKuB1wkvs0PUDLwSRdRXSyANbgogDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"27f18d9de28eaaad980481ff8166d361906e8dfbaa704a5c1a7cae289da77f59","last_reissued_at":"2026-07-05T11:12:02.133240Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:12:02.133240Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Diffusion Guidance Is a Controllable Policy Improvement Operator","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Kevin Frans, Pieter Abbeel, Seohong Park, Sergey Levine","submitted_at":"2025-05-29T14:06:50Z","abstract_excerpt":"At the core of reinforcement learning is the idea of learning beyond the performance in the data. However, scaling such systems has proven notoriously tricky. In contrast, techniques from generative modeling have proven remarkably scalable and are simple to train. In this work, we combine these strengths, by deriving a direct relation between policy improvement and guidance of diffusion models. The resulting framework, CFGRL, is trained with the simplicity of supervised learning, yet can further improve on the policies in the data. On offline RL tasks, we observe a reliable trend -- increased "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.23458","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.23458/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.23458","created_at":"2026-07-05T11:12:02.133323+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.23458v1","created_at":"2026-07-05T11:12:02.133323+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.23458","created_at":"2026-07-05T11:12:02.133323+00:00"},{"alias_kind":"pith_short_12","alias_value":"E7YY3HPCR2VK","created_at":"2026-07-05T11:12:02.133323+00:00"},{"alias_kind":"pith_short_16","alias_value":"E7YY3HPCR2VK3GAE","created_at":"2026-07-05T11:12:02.133323+00:00"},{"alias_kind":"pith_short_8","alias_value":"E7YY3HPC","created_at":"2026-07-05T11:12:02.133323+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":22,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.24231","citing_title":"FlowR2A: Learning Reward-to-Action Distribution for Multimodal Driving Planning","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2606.21406","citing_title":"Robot Self-Improvement via Human-Video Dynamics Models","ref_index":56,"is_internal_anchor":false},{"citing_arxiv_id":"2606.17551","citing_title":"Reversal Q-Learning","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2606.13675","citing_title":"Improving Robotic Generalist Policies via Flow Reversal Steering","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2607.02496","citing_title":"Controllable Sim Agents with Behavior Latents","ref_index":94,"is_internal_anchor":false},{"citing_arxiv_id":"2606.11087","citing_title":"Test-Time Gradient Guidance of Flow Policies in Reinforcement Learning","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2606.09615","citing_title":"DexPIE: Stable Dexterous Policy Improvement from Real-World Experience","ref_index":44,"is_internal_anchor":false},{"citing_arxiv_id":"2606.09009","citing_title":"Scaling by Diversified Experience for Vision-Language-Action Models","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2605.13435","citing_title":"Q-Flow: Stable and Expressive Reinforcement Learning with Flow-Based Policy","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2606.28939","citing_title":"ReGuide: From Test-Time Guidance to Self-Improving Diffusion Policies","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2606.29834","citing_title":"STEAM: Self-Supervised Temporal Ensemble Advantage Modeling for Real-World Robot Learning","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2602.11075","citing_title":"RISE: Self-Improving Robot Policy with Compositional World Model","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2602.13193","citing_title":"Steerable Vision-Language-Action Policies for Embodied Reasoning and Hierarchical Control","ref_index":67,"is_internal_anchor":false},{"citing_arxiv_id":"2605.13013","citing_title":"JEDI: Joint Embedding Diffusion World Model for Online Model-Based Reinforcement Learning","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2605.13435","citing_title":"Q-Flow: Stable and Expressive Reinforcement Learning with Flow-Based Policy","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2509.16117","citing_title":"DiffusionNFT: Online Diffusion Reinforcement with Forward Process","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2511.14759","citing_title":"$\\pi^{*}_{0.6}$: a VLA That Learns From Experience","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2605.03075","citing_title":"Refining Compositional Diffusion for Reliable Long-Horizon Planning","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2604.08174","citing_title":"Value-Guidance MeanFlow for Offline Multi-Agent Reinforcement Learning","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2604.08168","citing_title":"ViVa: A Video-Generative Value Model for Robot Reinforcement Learning","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2604.14265","citing_title":"Reinforcement Learning via Value Gradient Flow","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2604.15577","citing_title":"Reward Weighted Classifier-Free Guidance as Policy Improvement in Autoregressive Models","ref_index":4,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/E7YY3HPCR2VK3GAEQH7YCZWTMG","json":"https://pith.science/pith/E7YY3HPCR2VK3GAEQH7YCZWTMG.json","graph_json":"https://pith.science/api/pith-number/E7YY3HPCR2VK3GAEQH7YCZWTMG/graph.json","events_json":"https://pith.science/api/pith-number/E7YY3HPCR2VK3GAEQH7YCZWTMG/events.json","paper":"https://pith.science/paper/E7YY3HPC"},"agent_actions":{"view_html":"https://pith.science/pith/E7YY3HPCR2VK3GAEQH7YCZWTMG","download_json":"https://pith.science/pith/E7YY3HPCR2VK3GAEQH7YCZWTMG.json","view_paper":"https://pith.science/paper/E7YY3HPC","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.23458&json=true","fetch_graph":"https://pith.science/api/pith-number/E7YY3HPCR2VK3GAEQH7YCZWTMG/graph.json","fetch_events":"https://pith.science/api/pith-number/E7YY3HPCR2VK3GAEQH7YCZWTMG/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/E7YY3HPCR2VK3GAEQH7YCZWTMG/action/timestamp_anchor","attest_storage":"https://pith.science/pith/E7YY3HPCR2VK3GAEQH7YCZWTMG/action/storage_attestation","attest_author":"https://pith.science/pith/E7YY3HPCR2VK3GAEQH7YCZWTMG/action/author_attestation","sign_citation":"https://pith.science/pith/E7YY3HPCR2VK3GAEQH7YCZWTMG/action/citation_signature","submit_replication":"https://pith.science/pith/E7YY3HPCR2VK3GAEQH7YCZWTMG/action/replication_record"}},"created_at":"2026-07-05T11:12:02.133323+00:00","updated_at":"2026-07-05T11:12:02.133323+00:00"}