{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:NW6OMQRBQ6KQWL7ZNLTZAKBSPF","short_pith_number":"pith:NW6OMQRB","schema_version":"1.0","canonical_sha256":"6dbce6422187950b2ff96ae7902832794bf692a8d4d8c17bdaeb4803c57db44d","source":{"kind":"arxiv","id":"2310.07297","version":3},"attestation_state":"computed","paper":{"title":"Score Regularized Policy Optimization through Diffusion Behavior","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Cheng Lu, Hang Su, Huayu Chen, Jun Zhu, Zhengyi Wang","submitted_at":"2023-10-11T08:31:26Z","abstract_excerpt":"Recent developments in offline reinforcement learning have uncovered the immense potential of diffusion modeling, which excels at representing heterogeneous behavior policies. However, sampling from diffusion policies is considerably slow because it necessitates tens to hundreds of iterative inference steps for one action. To address this issue, we propose to extract an efficient deterministic inference policy from critic models and pretrained diffusion behavior models, leveraging the latter to directly regularize the policy gradient with the behavior distribution's score function during optim"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2310.07297","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2023-10-11T08:31:26Z","cross_cats_sorted":[],"title_canon_sha256":"04b07219c12228259ce30d056951688d08fea9676abfa959211e55d25d89e4c4","abstract_canon_sha256":"e0f96e5a4b582ae4632ece97c67c07713f61e79d687617c559f876817ba1d930"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:56:17.552270Z","signature_b64":"ADyJpr5U4IKGebScjLEIHamrSp7mg8ZQTmAGSj/1E/MphoHH1CHOI02N5WiNHIQd1ltiHhXiBFLW5k8/N3gfAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"6dbce6422187950b2ff96ae7902832794bf692a8d4d8c17bdaeb4803c57db44d","last_reissued_at":"2026-07-05T07:56:17.551819Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:56:17.551819Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Score Regularized Policy Optimization through Diffusion Behavior","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Cheng Lu, Hang Su, Huayu Chen, Jun Zhu, Zhengyi Wang","submitted_at":"2023-10-11T08:31:26Z","abstract_excerpt":"Recent developments in offline reinforcement learning have uncovered the immense potential of diffusion modeling, which excels at representing heterogeneous behavior policies. However, sampling from diffusion policies is considerably slow because it necessitates tens to hundreds of iterative inference steps for one action. To address this issue, we propose to extract an efficient deterministic inference policy from critic models and pretrained diffusion behavior models, leveraging the latter to directly regularize the policy gradient with the behavior distribution's score function during optim"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2310.07297","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2310.07297/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2310.07297","created_at":"2026-07-05T07:56:17.551882+00:00"},{"alias_kind":"arxiv_version","alias_value":"2310.07297v3","created_at":"2026-07-05T07:56:17.551882+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2310.07297","created_at":"2026-07-05T07:56:17.551882+00:00"},{"alias_kind":"pith_short_12","alias_value":"NW6OMQRBQ6KQ","created_at":"2026-07-05T07:56:17.551882+00:00"},{"alias_kind":"pith_short_16","alias_value":"NW6OMQRBQ6KQWL7Z","created_at":"2026-07-05T07:56:17.551882+00:00"},{"alias_kind":"pith_short_8","alias_value":"NW6OMQRB","created_at":"2026-07-05T07:56:17.551882+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":8,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.10613","citing_title":"Fast and Highly Expressive Policy Learning for Offline Reinforcement Learning via Bootstrapped Flow Q-Learning","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2605.27877","citing_title":"SPAR: Support-Preserving Action Rectification","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2605.29398","citing_title":"GDSD: Reinforcement Learning as Guided Denoiser Self-Distillation for Diffusion Language Models","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2506.15799","citing_title":"Steering Your Diffusion Policy with Latent Space Reinforcement Learning","ref_index":58,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08202","citing_title":"Beyond Penalization: Diffusion-based Out-of-Distribution Detection and Selective Regularization in Offline Reinforcement Learning","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08253","citing_title":"Path-Coupled Bellman Flows for Distributional Reinforcement Learning","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01663","citing_title":"Towards Efficient and Expressive Offline RL via Flow-Anchored Noise-conditioned Q-Learning","ref_index":55,"is_internal_anchor":false},{"citing_arxiv_id":"2604.17919","citing_title":"Fisher Decorator: Refining Flow Policy via a Local Transport Map","ref_index":7,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/NW6OMQRBQ6KQWL7ZNLTZAKBSPF","json":"https://pith.science/pith/NW6OMQRBQ6KQWL7ZNLTZAKBSPF.json","graph_json":"https://pith.science/api/pith-number/NW6OMQRBQ6KQWL7ZNLTZAKBSPF/graph.json","events_json":"https://pith.science/api/pith-number/NW6OMQRBQ6KQWL7ZNLTZAKBSPF/events.json","paper":"https://pith.science/paper/NW6OMQRB"},"agent_actions":{"view_html":"https://pith.science/pith/NW6OMQRBQ6KQWL7ZNLTZAKBSPF","download_json":"https://pith.science/pith/NW6OMQRBQ6KQWL7ZNLTZAKBSPF.json","view_paper":"https://pith.science/paper/NW6OMQRB","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2310.07297&json=true","fetch_graph":"https://pith.science/api/pith-number/NW6OMQRBQ6KQWL7ZNLTZAKBSPF/graph.json","fetch_events":"https://pith.science/api/pith-number/NW6OMQRBQ6KQWL7ZNLTZAKBSPF/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/NW6OMQRBQ6KQWL7ZNLTZAKBSPF/action/timestamp_anchor","attest_storage":"https://pith.science/pith/NW6OMQRBQ6KQWL7ZNLTZAKBSPF/action/storage_attestation","attest_author":"https://pith.science/pith/NW6OMQRBQ6KQWL7ZNLTZAKBSPF/action/author_attestation","sign_citation":"https://pith.science/pith/NW6OMQRBQ6KQWL7ZNLTZAKBSPF/action/citation_signature","submit_replication":"https://pith.science/pith/NW6OMQRBQ6KQWL7ZNLTZAKBSPF/action/replication_record"}},"created_at":"2026-07-05T07:56:17.551882+00:00","updated_at":"2026-07-05T07:56:17.551882+00:00"}