{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:7CDRWVVI7ZNHK6RB2ZELKZZ7TK","short_pith_number":"pith:7CDRWVVI","schema_version":"1.0","canonical_sha256":"f8871b56a8fe5a757a21d648b5673f9a95ddb03bb226f2358bfc5b0efc88931d","source":{"kind":"arxiv","id":"2305.13122","version":1},"attestation_state":"computed","paper":{"title":"Policy Representation via Diffusion Probability Model for Reinforcement Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Binbin Zhou, Cong Fang, Fenghao Lei, Long Yang, Shiting Wen, Yiming Yang, Yucun Zhong, Zhixiong Huang, Zhouchen Lin","submitted_at":"2023-05-22T15:23:41Z","abstract_excerpt":"Popular reinforcement learning (RL) algorithms tend to produce a unimodal policy distribution, which weakens the expressiveness of complicated policy and decays the ability of exploration. The diffusion probability model is powerful to learn complicated multimodal distributions, which has shown promising and potential applications to RL. In this paper, we formally build a theoretical foundation of policy representation via the diffusion probability model and provide practical implementations of diffusion policy for online model-free RL. Concretely, we character diffusion policy as a stochastic"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2305.13122","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2023-05-22T15:23:41Z","cross_cats_sorted":[],"title_canon_sha256":"46e5937315570c298ba02481855946bb1f0c770be826ab280ff9f34fdd9c593f","abstract_canon_sha256":"625930b9f3b19901cd17a98b62f2e53c42c6a1f849ff51b22ad1f442114cc7ff"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:12:26.618453Z","signature_b64":"qd2a6tLaWB2F2Z2BcEfhQMOx+SzhKlkSqfs6g2yRXW1+DIp1lieh+M7gbDL39nDP/+8PQ1q8hgFMOX7y2eOuDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f8871b56a8fe5a757a21d648b5673f9a95ddb03bb226f2358bfc5b0efc88931d","last_reissued_at":"2026-07-05T06:12:26.617950Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:12:26.617950Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Policy Representation via Diffusion Probability Model for Reinforcement Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Binbin Zhou, Cong Fang, Fenghao Lei, Long Yang, Shiting Wen, Yiming Yang, Yucun Zhong, Zhixiong Huang, Zhouchen Lin","submitted_at":"2023-05-22T15:23:41Z","abstract_excerpt":"Popular reinforcement learning (RL) algorithms tend to produce a unimodal policy distribution, which weakens the expressiveness of complicated policy and decays the ability of exploration. The diffusion probability model is powerful to learn complicated multimodal distributions, which has shown promising and potential applications to RL. In this paper, we formally build a theoretical foundation of policy representation via the diffusion probability model and provide practical implementations of diffusion policy for online model-free RL. Concretely, we character diffusion policy as a stochastic"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2305.13122","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2305.13122/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2305.13122","created_at":"2026-07-05T06:12:26.618007+00:00"},{"alias_kind":"arxiv_version","alias_value":"2305.13122v1","created_at":"2026-07-05T06:12:26.618007+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2305.13122","created_at":"2026-07-05T06:12:26.618007+00:00"},{"alias_kind":"pith_short_12","alias_value":"7CDRWVVI7ZNH","created_at":"2026-07-05T06:12:26.618007+00:00"},{"alias_kind":"pith_short_16","alias_value":"7CDRWVVI7ZNHK6RB","created_at":"2026-07-05T06:12:26.618007+00:00"},{"alias_kind":"pith_short_8","alias_value":"7CDRWVVI","created_at":"2026-07-05T06:12:26.618007+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":26,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.07967","citing_title":"Expressivity and Statistical Trade-offs in Diffusion Policy Learning","ref_index":65,"is_internal_anchor":true},{"citing_arxiv_id":"2606.17551","citing_title":"Reversal Q-Learning","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2606.11087","citing_title":"Test-Time Gradient Guidance of Flow Policies in Reinforcement Learning","ref_index":68,"is_internal_anchor":false},{"citing_arxiv_id":"2606.08015","citing_title":"Q-VGM: Q-Value-Gradient Matching for Off-Policy Reinforcement Learning of Flow-Matching VLA","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2606.06967","citing_title":"GenPO++: Generative Policy Optimization with Jacobian-free Likelihood Ratios","ref_index":48,"is_internal_anchor":false},{"citing_arxiv_id":"2605.25477","citing_title":"EXPO-FT: Sample-Efficient Reinforcement Learning Finetuning for Vision-Language-Action Models","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2605.26478","citing_title":"Efficient On-policy Visual-RL via Stochastic Decoupled Policy Gradient","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2605.30056","citing_title":"Sample-Efficient Diffusion-based Reinforcement Learning with Critic Guidance","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2605.30749","citing_title":"FLAG: Flow Policy MaxEnt-RL by Latent Augmented Guidance","ref_index":49,"is_internal_anchor":false},{"citing_arxiv_id":"2510.03508","citing_title":"D2 Actor Critic: Diffusion Actor Meets Distributional Critic","ref_index":37,"is_internal_anchor":false},{"citing_arxiv_id":"2605.23365","citing_title":"Score-Based One-step MeanFlow Policy Optimization","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2503.13934","citing_title":"COLSON: Controllable Learning-Based Social Navigation via Diffusion-Based Reinforcement Learning","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2605.21282","citing_title":"Stochastic MeanFlow Policies: One-Step Generative Control with Entropic Mirror Descent","ref_index":57,"is_internal_anchor":false},{"citing_arxiv_id":"2605.21282","citing_title":"Stochastic MeanFlow Policies: One-Step Generative Control with Entropic Mirror Descent","ref_index":57,"is_internal_anchor":false},{"citing_arxiv_id":"2605.16520","citing_title":"Global Convergence of Sampling-Based Nonconvex Optimization through Diffusion-Style Smoothing","ref_index":230,"is_internal_anchor":false},{"citing_arxiv_id":"2505.18780","citing_title":"DreamPolicy: A Unified World-model Policy for Scalable Humanoid Locomotion","ref_index":60,"is_internal_anchor":false},{"citing_arxiv_id":"2506.15799","citing_title":"Steering Your Diffusion Policy with Latent Space Reinforcement Learning","ref_index":62,"is_internal_anchor":false},{"citing_arxiv_id":"2409.00588","citing_title":"Diffusion Policy Policy Optimization","ref_index":103,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12416","citing_title":"Aligning Flow Map Policies with Optimal Q-Guidance","ref_index":46,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11387","citing_title":"Behavioral Mode Discovery for Fine-tuning Multimodal Generative Policies","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2604.19677","citing_title":"Learning Hybrid-Control Policies for High-Precision In-Contact Manipulation Under Uncertainty","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2604.09159","citing_title":"Truncated Rectified Flow Policy for Reinforcement Learning with One-Step Sampling","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07101","citing_title":"Decentralized Diffusion Policy Learning for Enhanced Exploration in Cooperative Multi-agent Reinforcement Learning","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2604.14265","citing_title":"Reinforcement Learning via Value Gradient Flow","ref_index":76,"is_internal_anchor":false},{"citing_arxiv_id":"2604.14698","citing_title":"Mean Flow Policy Optimization","ref_index":6,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/7CDRWVVI7ZNHK6RB2ZELKZZ7TK","json":"https://pith.science/pith/7CDRWVVI7ZNHK6RB2ZELKZZ7TK.json","graph_json":"https://pith.science/api/pith-number/7CDRWVVI7ZNHK6RB2ZELKZZ7TK/graph.json","events_json":"https://pith.science/api/pith-number/7CDRWVVI7ZNHK6RB2ZELKZZ7TK/events.json","paper":"https://pith.science/paper/7CDRWVVI"},"agent_actions":{"view_html":"https://pith.science/pith/7CDRWVVI7ZNHK6RB2ZELKZZ7TK","download_json":"https://pith.science/pith/7CDRWVVI7ZNHK6RB2ZELKZZ7TK.json","view_paper":"https://pith.science/paper/7CDRWVVI","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2305.13122&json=true","fetch_graph":"https://pith.science/api/pith-number/7CDRWVVI7ZNHK6RB2ZELKZZ7TK/graph.json","fetch_events":"https://pith.science/api/pith-number/7CDRWVVI7ZNHK6RB2ZELKZZ7TK/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/7CDRWVVI7ZNHK6RB2ZELKZZ7TK/action/timestamp_anchor","attest_storage":"https://pith.science/pith/7CDRWVVI7ZNHK6RB2ZELKZZ7TK/action/storage_attestation","attest_author":"https://pith.science/pith/7CDRWVVI7ZNHK6RB2ZELKZZ7TK/action/author_attestation","sign_citation":"https://pith.science/pith/7CDRWVVI7ZNHK6RB2ZELKZZ7TK/action/citation_signature","submit_replication":"https://pith.science/pith/7CDRWVVI7ZNHK6RB2ZELKZZ7TK/action/replication_record"}},"created_at":"2026-07-05T06:12:26.618007+00:00","updated_at":"2026-07-05T06:12:26.618007+00:00"}