{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:HHNR2WVKBTWUS64LMWY7T2P2ED","short_pith_number":"pith:HHNR2WVK","schema_version":"1.0","canonical_sha256":"39db1d5aaa0ced497b8b65b1f9e9fa20c8a9410f7f99aea59192a6622e113839","source":{"kind":"arxiv","id":"2309.16984","version":2},"attestation_state":"computed","paper":{"title":"Consistency Models as a Rich and Efficient Policy Class for Reinforcement Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Chi Jin, Zihan Ding","submitted_at":"2023-09-29T05:05:54Z","abstract_excerpt":"Score-based generative models like the diffusion model have been testified to be effective in modeling multi-modal data from image generation to reinforcement learning (RL). However, the inference process of diffusion model can be slow, which hinders its usage in RL with iterative sampling. We propose to apply the consistency model as an efficient yet expressive policy representation, namely consistency policy, with an actor-critic style algorithm for three typical RL settings: offline, offline-to-online and online. For offline RL, we demonstrate the expressiveness of generative models as poli"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2309.16984","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2023-09-29T05:05:54Z","cross_cats_sorted":[],"title_canon_sha256":"4b0efd49ad135ef085793a339d4f12c2f14c98931a1c4c71172cccdf47c457c5","abstract_canon_sha256":"67195acdb9c4d460dce5c7845bd73f6f7c8b225118231675126631631ca62919"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:56:17.217042Z","signature_b64":"9OKpPeslSpU1dt85uDfTudPR7kzkM1jLYevLLZzJL2kgdEOkhf7LKBeqo/Z/brX2WjX0n+8VxWZNYByHZzTBAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"39db1d5aaa0ced497b8b65b1f9e9fa20c8a9410f7f99aea59192a6622e113839","last_reissued_at":"2026-07-05T07:56:17.216601Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:56:17.216601Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Consistency Models as a Rich and Efficient Policy Class for Reinforcement Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Chi Jin, Zihan Ding","submitted_at":"2023-09-29T05:05:54Z","abstract_excerpt":"Score-based generative models like the diffusion model have been testified to be effective in modeling multi-modal data from image generation to reinforcement learning (RL). However, the inference process of diffusion model can be slow, which hinders its usage in RL with iterative sampling. We propose to apply the consistency model as an efficient yet expressive policy representation, namely consistency policy, with an actor-critic style algorithm for three typical RL settings: offline, offline-to-online and online. For offline RL, we demonstrate the expressiveness of generative models as poli"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2309.16984","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2309.16984/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2309.16984","created_at":"2026-07-05T07:56:17.216660+00:00"},{"alias_kind":"arxiv_version","alias_value":"2309.16984v2","created_at":"2026-07-05T07:56:17.216660+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2309.16984","created_at":"2026-07-05T07:56:17.216660+00:00"},{"alias_kind":"pith_short_12","alias_value":"HHNR2WVKBTWU","created_at":"2026-07-05T07:56:17.216660+00:00"},{"alias_kind":"pith_short_16","alias_value":"HHNR2WVKBTWUS64L","created_at":"2026-07-05T07:56:17.216660+00:00"},{"alias_kind":"pith_short_8","alias_value":"HHNR2WVK","created_at":"2026-07-05T07:56:17.216660+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":9,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.10613","citing_title":"Fast and Highly Expressive Policy Learning for Offline Reinforcement Learning via Bootstrapped Flow Q-Learning","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2606.11087","citing_title":"Test-Time Gradient Guidance of Flow Policies in Reinforcement Learning","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2606.08602","citing_title":"Reinforcement Learning for Flow-Matching Policies with Density Transport","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2506.15799","citing_title":"Steering Your Diffusion Policy with Latent Space Reinforcement Learning","ref_index":48,"is_internal_anchor":false},{"citing_arxiv_id":"2409.00588","citing_title":"Diffusion Policy Policy Optimization","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12416","citing_title":"Aligning Flow Map Policies with Optimal Q-Guidance","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01663","citing_title":"Towards Efficient and Expressive Offline RL via Flow-Anchored Noise-conditioned Q-Learning","ref_index":43,"is_internal_anchor":false},{"citing_arxiv_id":"2604.14265","citing_title":"Reinforcement Learning via Value Gradient Flow","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2604.17919","citing_title":"Fisher Decorator: Refining Flow Policy via a Local Transport Map","ref_index":14,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/HHNR2WVKBTWUS64LMWY7T2P2ED","json":"https://pith.science/pith/HHNR2WVKBTWUS64LMWY7T2P2ED.json","graph_json":"https://pith.science/api/pith-number/HHNR2WVKBTWUS64LMWY7T2P2ED/graph.json","events_json":"https://pith.science/api/pith-number/HHNR2WVKBTWUS64LMWY7T2P2ED/events.json","paper":"https://pith.science/paper/HHNR2WVK"},"agent_actions":{"view_html":"https://pith.science/pith/HHNR2WVKBTWUS64LMWY7T2P2ED","download_json":"https://pith.science/pith/HHNR2WVKBTWUS64LMWY7T2P2ED.json","view_paper":"https://pith.science/paper/HHNR2WVK","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2309.16984&json=true","fetch_graph":"https://pith.science/api/pith-number/HHNR2WVKBTWUS64LMWY7T2P2ED/graph.json","fetch_events":"https://pith.science/api/pith-number/HHNR2WVKBTWUS64LMWY7T2P2ED/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/HHNR2WVKBTWUS64LMWY7T2P2ED/action/timestamp_anchor","attest_storage":"https://pith.science/pith/HHNR2WVKBTWUS64LMWY7T2P2ED/action/storage_attestation","attest_author":"https://pith.science/pith/HHNR2WVKBTWUS64LMWY7T2P2ED/action/author_attestation","sign_citation":"https://pith.science/pith/HHNR2WVKBTWUS64LMWY7T2P2ED/action/citation_signature","submit_replication":"https://pith.science/pith/HHNR2WVKBTWUS64LMWY7T2P2ED/action/replication_record"}},"created_at":"2026-07-05T07:56:17.216660+00:00","updated_at":"2026-07-05T07:56:17.216660+00:00"}