{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:KPSUWKPRIT472FD2HPZOIJUJQA","short_pith_number":"pith:KPSUWKPR","schema_version":"1.0","canonical_sha256":"53e54b29f144f9fd147a3bf2e4268980070edb3ac08fdd1707af79b6fecf1c06","source":{"kind":"arxiv","id":"2405.16173","version":3},"attestation_state":"computed","paper":{"title":"Diffusion-based Reinforcement Learning via Q-weighted Variational Policy Optimization","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Jingya Wang, Jingyi Yu, Kan Ren, Ke Hu, Shutong Ding, Weinan Zhang, Ye Shi, Zhenhao Zhang","submitted_at":"2024-05-25T10:45:46Z","abstract_excerpt":"Diffusion models have garnered widespread attention in Reinforcement Learning (RL) for their powerful expressiveness and multimodality. It has been verified that utilizing diffusion policies can significantly improve the performance of RL algorithms in continuous control tasks by overcoming the limitations of unimodal policies, such as Gaussian policies, and providing the agent with enhanced exploration capabilities. However, existing works mainly focus on the application of diffusion policies in offline RL, while their incorporation into online RL is less investigated. The training objective "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2405.16173","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-05-25T10:45:46Z","cross_cats_sorted":[],"title_canon_sha256":"66fbafec92d898a874dda4f6a5789ddfdb905e68ee073875b096cc7a31c0f9a3","abstract_canon_sha256":"d30d10b728f02345bd2b276d2b19e89ea919d419df38d74a6cc3a4b446445e5b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:49:42.842180Z","signature_b64":"cbLX8EESnAJQ/1aMVT2arlZSAVpfzfuHNzVRnU6JJhYbLqC61KF9/rzFiSrL8zEYppIy3A8uLZh3F84FN1zzDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"53e54b29f144f9fd147a3bf2e4268980070edb3ac08fdd1707af79b6fecf1c06","last_reissued_at":"2026-07-05T09:49:42.841696Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:49:42.841696Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Diffusion-based Reinforcement Learning via Q-weighted Variational Policy Optimization","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Jingya Wang, Jingyi Yu, Kan Ren, Ke Hu, Shutong Ding, Weinan Zhang, Ye Shi, Zhenhao Zhang","submitted_at":"2024-05-25T10:45:46Z","abstract_excerpt":"Diffusion models have garnered widespread attention in Reinforcement Learning (RL) for their powerful expressiveness and multimodality. It has been verified that utilizing diffusion policies can significantly improve the performance of RL algorithms in continuous control tasks by overcoming the limitations of unimodal policies, such as Gaussian policies, and providing the agent with enhanced exploration capabilities. However, existing works mainly focus on the application of diffusion policies in offline RL, while their incorporation into online RL is less investigated. The training objective "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2405.16173","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2405.16173/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2405.16173","created_at":"2026-07-05T09:49:42.841753+00:00"},{"alias_kind":"arxiv_version","alias_value":"2405.16173v3","created_at":"2026-07-05T09:49:42.841753+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2405.16173","created_at":"2026-07-05T09:49:42.841753+00:00"},{"alias_kind":"pith_short_12","alias_value":"KPSUWKPRIT47","created_at":"2026-07-05T09:49:42.841753+00:00"},{"alias_kind":"pith_short_16","alias_value":"KPSUWKPRIT472FD2","created_at":"2026-07-05T09:49:42.841753+00:00"},{"alias_kind":"pith_short_8","alias_value":"KPSUWKPR","created_at":"2026-07-05T09:49:42.841753+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":8,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.06967","citing_title":"GenPO++: Generative Policy Optimization with Jacobian-free Likelihood Ratios","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2606.05407","citing_title":"MoDex: A Diffusion Policy for Sequential Multi-Object Dexterous Grasping","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2509.22963","citing_title":"Reinforcement Learning with Discrete Diffusion Policies for Combinatorial Action Spaces","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2605.16520","citing_title":"Global Convergence of Sampling-Based Nonconvex Optimization through Diffusion-Style Smoothing","ref_index":57,"is_internal_anchor":false},{"citing_arxiv_id":"2507.07986","citing_title":"EXPO: Stable Reinforcement Learning with Expressive Policies","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2506.15799","citing_title":"Steering Your Diffusion Policy with Latent Space Reinforcement Learning","ref_index":45,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11726","citing_title":"Block-R1: Rethinking the Role of Block Size in Multi-domain Reinforcement Learning for Diffusion Large Language Models","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11726","citing_title":"Block-R1: Rethinking the Role of Block Size in Multi-domain Reinforcement Learning for Diffusion Large Language Models","ref_index":16,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/KPSUWKPRIT472FD2HPZOIJUJQA","json":"https://pith.science/pith/KPSUWKPRIT472FD2HPZOIJUJQA.json","graph_json":"https://pith.science/api/pith-number/KPSUWKPRIT472FD2HPZOIJUJQA/graph.json","events_json":"https://pith.science/api/pith-number/KPSUWKPRIT472FD2HPZOIJUJQA/events.json","paper":"https://pith.science/paper/KPSUWKPR"},"agent_actions":{"view_html":"https://pith.science/pith/KPSUWKPRIT472FD2HPZOIJUJQA","download_json":"https://pith.science/pith/KPSUWKPRIT472FD2HPZOIJUJQA.json","view_paper":"https://pith.science/paper/KPSUWKPR","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2405.16173&json=true","fetch_graph":"https://pith.science/api/pith-number/KPSUWKPRIT472FD2HPZOIJUJQA/graph.json","fetch_events":"https://pith.science/api/pith-number/KPSUWKPRIT472FD2HPZOIJUJQA/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/KPSUWKPRIT472FD2HPZOIJUJQA/action/timestamp_anchor","attest_storage":"https://pith.science/pith/KPSUWKPRIT472FD2HPZOIJUJQA/action/storage_attestation","attest_author":"https://pith.science/pith/KPSUWKPRIT472FD2HPZOIJUJQA/action/author_attestation","sign_citation":"https://pith.science/pith/KPSUWKPRIT472FD2HPZOIJUJQA/action/citation_signature","submit_replication":"https://pith.science/pith/KPSUWKPRIT472FD2HPZOIJUJQA/action/replication_record"}},"created_at":"2026-07-05T09:49:42.841753+00:00","updated_at":"2026-07-05T09:49:42.841753+00:00"}