{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:TLZAFEVTM3KYNGP3PBOSIP6N6Y","short_pith_number":"pith:TLZAFEVT","schema_version":"1.0","canonical_sha256":"9af20292b366d58699fb785d243fcdf62b3cea6b1b57f25c3565f35406dff40b","source":{"kind":"arxiv","id":"2306.06329","version":1},"attestation_state":"computed","paper":{"title":"HIPODE: Enhancing Offline Reinforcement Learning with High-Quality Synthetic Data from a Policy-Decoupled Approach","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Jinyi Liu, Shixi Lian, Yan Zheng, Yi Ma, Zhaopeng Meng","submitted_at":"2023-06-10T01:49:01Z","abstract_excerpt":"Offline reinforcement learning (ORL) has gained attention as a means of training reinforcement learning models using pre-collected static data. To address the issue of limited data and improve downstream ORL performance, recent work has attempted to expand the dataset's coverage through data augmentation. However, most of these methods are tied to a specific policy (policy-dependent), where the generated data can only guarantee to support the current downstream ORL policy, limiting its usage scope on other downstream policies. Moreover, the quality of synthetic data is often not well-controlle"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2306.06329","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2023-06-10T01:49:01Z","cross_cats_sorted":[],"title_canon_sha256":"6c56fb650b98d4164487169e01ef768a53ea0522f2d1b643de11169b4babdbab","abstract_canon_sha256":"743bc691c7f768851b33f5a103fbacef1671bac7d6224ca18a30abcbbda37576"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:19:23.491279Z","signature_b64":"7kyCIBwrYF6zn1bEeD24C3EenQj82gL4i+CyJvPlUuGRQIQaYGcTNAKLKhioajjcZXW7u9uQPWnSkXd1TmcuDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"9af20292b366d58699fb785d243fcdf62b3cea6b1b57f25c3565f35406dff40b","last_reissued_at":"2026-07-05T06:19:23.490852Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:19:23.490852Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"HIPODE: Enhancing Offline Reinforcement Learning with High-Quality Synthetic Data from a Policy-Decoupled Approach","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Jinyi Liu, Shixi Lian, Yan Zheng, Yi Ma, Zhaopeng Meng","submitted_at":"2023-06-10T01:49:01Z","abstract_excerpt":"Offline reinforcement learning (ORL) has gained attention as a means of training reinforcement learning models using pre-collected static data. To address the issue of limited data and improve downstream ORL performance, recent work has attempted to expand the dataset's coverage through data augmentation. However, most of these methods are tied to a specific policy (policy-dependent), where the generated data can only guarantee to support the current downstream ORL policy, limiting its usage scope on other downstream policies. Moreover, the quality of synthetic data is often not well-controlle"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2306.06329","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2306.06329/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2306.06329","created_at":"2026-07-05T06:19:23.490908+00:00"},{"alias_kind":"arxiv_version","alias_value":"2306.06329v1","created_at":"2026-07-05T06:19:23.490908+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2306.06329","created_at":"2026-07-05T06:19:23.490908+00:00"},{"alias_kind":"pith_short_12","alias_value":"TLZAFEVTM3KY","created_at":"2026-07-05T06:19:23.490908+00:00"},{"alias_kind":"pith_short_16","alias_value":"TLZAFEVTM3KYNGP3","created_at":"2026-07-05T06:19:23.490908+00:00"},{"alias_kind":"pith_short_8","alias_value":"TLZAFEVT","created_at":"2026-07-05T06:19:23.490908+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2508.03194","citing_title":"Scaling DRL for Decision Making: A Survey on Data, Network, and Training Budget Strategies","ref_index":21,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/TLZAFEVTM3KYNGP3PBOSIP6N6Y","json":"https://pith.science/pith/TLZAFEVTM3KYNGP3PBOSIP6N6Y.json","graph_json":"https://pith.science/api/pith-number/TLZAFEVTM3KYNGP3PBOSIP6N6Y/graph.json","events_json":"https://pith.science/api/pith-number/TLZAFEVTM3KYNGP3PBOSIP6N6Y/events.json","paper":"https://pith.science/paper/TLZAFEVT"},"agent_actions":{"view_html":"https://pith.science/pith/TLZAFEVTM3KYNGP3PBOSIP6N6Y","download_json":"https://pith.science/pith/TLZAFEVTM3KYNGP3PBOSIP6N6Y.json","view_paper":"https://pith.science/paper/TLZAFEVT","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2306.06329&json=true","fetch_graph":"https://pith.science/api/pith-number/TLZAFEVTM3KYNGP3PBOSIP6N6Y/graph.json","fetch_events":"https://pith.science/api/pith-number/TLZAFEVTM3KYNGP3PBOSIP6N6Y/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/TLZAFEVTM3KYNGP3PBOSIP6N6Y/action/timestamp_anchor","attest_storage":"https://pith.science/pith/TLZAFEVTM3KYNGP3PBOSIP6N6Y/action/storage_attestation","attest_author":"https://pith.science/pith/TLZAFEVTM3KYNGP3PBOSIP6N6Y/action/author_attestation","sign_citation":"https://pith.science/pith/TLZAFEVTM3KYNGP3PBOSIP6N6Y/action/citation_signature","submit_replication":"https://pith.science/pith/TLZAFEVTM3KYNGP3PBOSIP6N6Y/action/replication_record"}},"created_at":"2026-07-05T06:19:23.490908+00:00","updated_at":"2026-07-05T06:19:23.490908+00:00"}