{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:JRDETDOHDFGG7SYPAEVNAYWAOT","short_pith_number":"pith:JRDETDOH","schema_version":"1.0","canonical_sha256":"4c46498dc7194c6fcb0f012ad062c074c722f8a44bbb8d7f1f97877df3786809","source":{"kind":"arxiv","id":"2405.19690","version":3},"attestation_state":"computed","paper":{"title":"Diffusion Policies creating a Trust Region for Offline Reinforcement Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Mingyuan Zhou, Tianyu Chen, Zhendong Wang","submitted_at":"2024-05-30T05:04:33Z","abstract_excerpt":"Offline reinforcement learning (RL) leverages pre-collected datasets to train optimal policies. Diffusion Q-Learning (DQL), introducing diffusion models as a powerful and expressive policy class, significantly boosts the performance of offline RL. However, its reliance on iterative denoising sampling to generate actions slows down both training and inference. While several recent attempts have tried to accelerate diffusion-QL, the improvement in training and/or inference speed often results in degraded performance. In this paper, we introduce a dual policy approach, Diffusion Trusted Q-Learnin"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2405.19690","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-05-30T05:04:33Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"476b158b69415f5677c82ee2c7ace70a0eb7b8bcb3e57e72aaf095b9dfbb45b0","abstract_canon_sha256":"5c9e0e6209a328e808188a7a9c5e8723e3aab84484067586004515f739a06447"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:29:29.438745Z","signature_b64":"4wbW/GQ1SidXwufj4RBqLeL03l+mmC9qbYKP2N0oZ5zYnGW70ZKKR7ZdYlVcitvmfQLP5OwxGBF/abB6QchmDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"4c46498dc7194c6fcb0f012ad062c074c722f8a44bbb8d7f1f97877df3786809","last_reissued_at":"2026-07-05T09:29:29.438233Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:29:29.438233Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Diffusion Policies creating a Trust Region for Offline Reinforcement Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Mingyuan Zhou, Tianyu Chen, Zhendong Wang","submitted_at":"2024-05-30T05:04:33Z","abstract_excerpt":"Offline reinforcement learning (RL) leverages pre-collected datasets to train optimal policies. Diffusion Q-Learning (DQL), introducing diffusion models as a powerful and expressive policy class, significantly boosts the performance of offline RL. However, its reliance on iterative denoising sampling to generate actions slows down both training and inference. While several recent attempts have tried to accelerate diffusion-QL, the improvement in training and/or inference speed often results in degraded performance. In this paper, we introduce a dual policy approach, Diffusion Trusted Q-Learnin"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2405.19690","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2405.19690/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2405.19690","created_at":"2026-07-05T09:29:29.438294+00:00"},{"alias_kind":"arxiv_version","alias_value":"2405.19690v3","created_at":"2026-07-05T09:29:29.438294+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2405.19690","created_at":"2026-07-05T09:29:29.438294+00:00"},{"alias_kind":"pith_short_12","alias_value":"JRDETDOHDFGG","created_at":"2026-07-05T09:29:29.438294+00:00"},{"alias_kind":"pith_short_16","alias_value":"JRDETDOHDFGG7SYP","created_at":"2026-07-05T09:29:29.438294+00:00"},{"alias_kind":"pith_short_8","alias_value":"JRDETDOH","created_at":"2026-07-05T09:29:29.438294+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2506.15799","citing_title":"Steering Your Diffusion Policy with Latent Space Reinforcement Learning","ref_index":57,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11387","citing_title":"Behavioral Mode Discovery for Fine-tuning Multimodal Generative Policies","ref_index":62,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/JRDETDOHDFGG7SYPAEVNAYWAOT","json":"https://pith.science/pith/JRDETDOHDFGG7SYPAEVNAYWAOT.json","graph_json":"https://pith.science/api/pith-number/JRDETDOHDFGG7SYPAEVNAYWAOT/graph.json","events_json":"https://pith.science/api/pith-number/JRDETDOHDFGG7SYPAEVNAYWAOT/events.json","paper":"https://pith.science/paper/JRDETDOH"},"agent_actions":{"view_html":"https://pith.science/pith/JRDETDOHDFGG7SYPAEVNAYWAOT","download_json":"https://pith.science/pith/JRDETDOHDFGG7SYPAEVNAYWAOT.json","view_paper":"https://pith.science/paper/JRDETDOH","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2405.19690&json=true","fetch_graph":"https://pith.science/api/pith-number/JRDETDOHDFGG7SYPAEVNAYWAOT/graph.json","fetch_events":"https://pith.science/api/pith-number/JRDETDOHDFGG7SYPAEVNAYWAOT/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/JRDETDOHDFGG7SYPAEVNAYWAOT/action/timestamp_anchor","attest_storage":"https://pith.science/pith/JRDETDOHDFGG7SYPAEVNAYWAOT/action/storage_attestation","attest_author":"https://pith.science/pith/JRDETDOHDFGG7SYPAEVNAYWAOT/action/author_attestation","sign_citation":"https://pith.science/pith/JRDETDOHDFGG7SYPAEVNAYWAOT/action/citation_signature","submit_replication":"https://pith.science/pith/JRDETDOHDFGG7SYPAEVNAYWAOT/action/replication_record"}},"created_at":"2026-07-05T09:29:29.438294+00:00","updated_at":"2026-07-05T09:29:29.438294+00:00"}