{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:7MFQWPCO7YR5VMRLK2XGSTHPY3","short_pith_number":"pith:7MFQWPCO","schema_version":"1.0","canonical_sha256":"fb0b0b3c4efe23dab22b56ae694cefc6f73668908d129f49f37f033d35ffe7c0","source":{"kind":"arxiv","id":"2209.14548","version":2},"attestation_state":"computed","paper":{"title":"Offline Reinforcement Learning via High-Fidelity Generative Behavior Modeling","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Cheng Lu, Chengyang Ying, Hang Su, Huayu Chen, Jun Zhu","submitted_at":"2022-09-29T04:36:23Z","abstract_excerpt":"In offline reinforcement learning, weighted regression is a common method to ensure the learned policy stays close to the behavior policy and to prevent selecting out-of-sample actions. In this work, we show that due to the limited distributional expressivity of policy models, previous methods might still select unseen actions during training, which deviates from their initial motivation. To address this problem, we adopt a generative approach by decoupling the learned policy into two parts: an expressive generative behavior model and an action evaluation model. The key insight is that such de"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2209.14548","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2022-09-29T04:36:23Z","cross_cats_sorted":[],"title_canon_sha256":"88abec480546c4288bea75eeb6c2369e61a9ab5d1085613612ba9e49c42270eb","abstract_canon_sha256":"0d490b003fa0a4c5c3410f02a203f03398a852d203bdbb59d0703e011318a912"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:46:24.676040Z","signature_b64":"TK2qoej+pUrvrc1EKGkGOCSS2byska9eBYelQnGgZHASXIOFSsm1Impq4lX9kV6Cc6F+um8M+AWLlBVJdQkpAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"fb0b0b3c4efe23dab22b56ae694cefc6f73668908d129f49f37f033d35ffe7c0","last_reissued_at":"2026-07-05T05:46:24.675535Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:46:24.675535Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Offline Reinforcement Learning via High-Fidelity Generative Behavior Modeling","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Cheng Lu, Chengyang Ying, Hang Su, Huayu Chen, Jun Zhu","submitted_at":"2022-09-29T04:36:23Z","abstract_excerpt":"In offline reinforcement learning, weighted regression is a common method to ensure the learned policy stays close to the behavior policy and to prevent selecting out-of-sample actions. In this work, we show that due to the limited distributional expressivity of policy models, previous methods might still select unseen actions during training, which deviates from their initial motivation. To address this problem, we adopt a generative approach by decoupling the learned policy into two parts: an expressive generative behavior model and an action evaluation model. The key insight is that such de"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2209.14548","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2209.14548/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2209.14548","created_at":"2026-07-05T05:46:24.675585+00:00"},{"alias_kind":"arxiv_version","alias_value":"2209.14548v2","created_at":"2026-07-05T05:46:24.675585+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2209.14548","created_at":"2026-07-05T05:46:24.675585+00:00"},{"alias_kind":"pith_short_12","alias_value":"7MFQWPCO7YR5","created_at":"2026-07-05T05:46:24.675585+00:00"},{"alias_kind":"pith_short_16","alias_value":"7MFQWPCO7YR5VMRL","created_at":"2026-07-05T05:46:24.675585+00:00"},{"alias_kind":"pith_short_8","alias_value":"7MFQWPCO","created_at":"2026-07-05T05:46:24.675585+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":12,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.11087","citing_title":"Test-Time Gradient Guidance of Flow Policies in Reinforcement Learning","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2606.06967","citing_title":"GenPO++: Generative Policy Optimization with Jacobian-free Likelihood Ratios","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2606.01151","citing_title":"Lagrangian Perturbation Diffusion Steering: Latent Reinforcement Learning for Generative Policies","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2511.03828","citing_title":"From Static Constraints to Dynamic Adaptation: Sample-Level Constraint Relaxation for Offline-to-Online Reinforcement Learning","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2507.07986","citing_title":"EXPO: Stable Reinforcement Learning with Expressive Policies","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2511.03828","citing_title":"From Static Constraints to Dynamic Adaptation: Sample-Level Constraint Relaxation for Offline-to-Online Reinforcement Learning","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2506.15799","citing_title":"Steering Your Diffusion Policy with Latent Space Reinforcement Learning","ref_index":51,"is_internal_anchor":false},{"citing_arxiv_id":"2409.00588","citing_title":"Diffusion Policy Policy Optimization","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2304.10573","citing_title":"IDQL: Implicit Q-Learning as an Actor-Critic Method with Diffusion Policies","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08202","citing_title":"Beyond Penalization: Diffusion-based Out-of-Distribution Detection and Selective Regularization in Offline Reinforcement Learning","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01663","citing_title":"Towards Efficient and Expressive Offline RL via Flow-Anchored Noise-conditioned Q-Learning","ref_index":39,"is_internal_anchor":false},{"citing_arxiv_id":"2604.19730","citing_title":"FASTER: Value-Guided Sampling for Fast RL","ref_index":8,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/7MFQWPCO7YR5VMRLK2XGSTHPY3","json":"https://pith.science/pith/7MFQWPCO7YR5VMRLK2XGSTHPY3.json","graph_json":"https://pith.science/api/pith-number/7MFQWPCO7YR5VMRLK2XGSTHPY3/graph.json","events_json":"https://pith.science/api/pith-number/7MFQWPCO7YR5VMRLK2XGSTHPY3/events.json","paper":"https://pith.science/paper/7MFQWPCO"},"agent_actions":{"view_html":"https://pith.science/pith/7MFQWPCO7YR5VMRLK2XGSTHPY3","download_json":"https://pith.science/pith/7MFQWPCO7YR5VMRLK2XGSTHPY3.json","view_paper":"https://pith.science/paper/7MFQWPCO","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2209.14548&json=true","fetch_graph":"https://pith.science/api/pith-number/7MFQWPCO7YR5VMRLK2XGSTHPY3/graph.json","fetch_events":"https://pith.science/api/pith-number/7MFQWPCO7YR5VMRLK2XGSTHPY3/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/7MFQWPCO7YR5VMRLK2XGSTHPY3/action/timestamp_anchor","attest_storage":"https://pith.science/pith/7MFQWPCO7YR5VMRLK2XGSTHPY3/action/storage_attestation","attest_author":"https://pith.science/pith/7MFQWPCO7YR5VMRLK2XGSTHPY3/action/author_attestation","sign_citation":"https://pith.science/pith/7MFQWPCO7YR5VMRLK2XGSTHPY3/action/citation_signature","submit_replication":"https://pith.science/pith/7MFQWPCO7YR5VMRLK2XGSTHPY3/action/replication_record"}},"created_at":"2026-07-05T05:46:24.675585+00:00","updated_at":"2026-07-05T05:46:24.675585+00:00"}