{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:HLBZF6WZC3QCI6XN2TQ4SEFPST","short_pith_number":"pith:HLBZF6WZ","schema_version":"1.0","canonical_sha256":"3ac392fad916e0247aedd4e1c910af94cfa49a991362a9361d8ca4372e3e696d","source":{"kind":"arxiv","id":"2110.01548","version":2},"attestation_state":"computed","paper":{"title":"Uncertainty-Based Offline Reinforcement Learning with Diversified Q-Ensemble","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Gaon An, Hyun Oh Song, Jang-Hyun Kim, Seungyong Moon","submitted_at":"2021-10-04T16:40:13Z","abstract_excerpt":"Offline reinforcement learning (offline RL), which aims to find an optimal policy from a previously collected static dataset, bears algorithmic difficulties due to function approximation errors from out-of-distribution (OOD) data points. To this end, offline RL algorithms adopt either a constraint or a penalty term that explicitly guides the policy to stay close to the given dataset. However, prior methods typically require accurate estimation of the behavior policy or sampling from OOD data points, which themselves can be a non-trivial problem. Moreover, these methods under-utilize the genera"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2110.01548","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2021-10-04T16:40:13Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"e2c884f8f36fbced71a4a15b4706ccd2a62e1acb4c6c2706c93c44c84dc71ee9","abstract_canon_sha256":"fe9a9f627591f10c98b0792ace00662e98a9b4bd6467474824520a6ea7afd053"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T03:20:04.260649Z","signature_b64":"uirNgS+fqHsTq6RVHxS0IHJX5ui58kecEiGNF9hLQtZeIYIZyPGkix4fXuSWKObmuhLZJq8YHbdIcqh44klUCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"3ac392fad916e0247aedd4e1c910af94cfa49a991362a9361d8ca4372e3e696d","last_reissued_at":"2026-07-05T03:20:04.260207Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T03:20:04.260207Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Uncertainty-Based Offline Reinforcement Learning with Diversified Q-Ensemble","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Gaon An, Hyun Oh Song, Jang-Hyun Kim, Seungyong Moon","submitted_at":"2021-10-04T16:40:13Z","abstract_excerpt":"Offline reinforcement learning (offline RL), which aims to find an optimal policy from a previously collected static dataset, bears algorithmic difficulties due to function approximation errors from out-of-distribution (OOD) data points. To this end, offline RL algorithms adopt either a constraint or a penalty term that explicitly guides the policy to stay close to the given dataset. However, prior methods typically require accurate estimation of the behavior policy or sampling from OOD data points, which themselves can be a non-trivial problem. Moreover, these methods under-utilize the genera"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2110.01548","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2110.01548/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2110.01548","created_at":"2026-07-05T03:20:04.260260+00:00"},{"alias_kind":"arxiv_version","alias_value":"2110.01548v2","created_at":"2026-07-05T03:20:04.260260+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2110.01548","created_at":"2026-07-05T03:20:04.260260+00:00"},{"alias_kind":"pith_short_12","alias_value":"HLBZF6WZC3QC","created_at":"2026-07-05T03:20:04.260260+00:00"},{"alias_kind":"pith_short_16","alias_value":"HLBZF6WZC3QCI6XN","created_at":"2026-07-05T03:20:04.260260+00:00"},{"alias_kind":"pith_short_8","alias_value":"HLBZF6WZ","created_at":"2026-07-05T03:20:04.260260+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2607.02092","citing_title":"Guided Action Flow: Q-Guided Inference for Flow-Matching Vision-Language-Action Policies","ref_index":46,"is_internal_anchor":false},{"citing_arxiv_id":"2605.22711","citing_title":"Abstraction for Offline Goal-Conditioned Reinforcement Learning","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11151","citing_title":"RankQ: Offline-to-Online Reinforcement Learning via Self-Supervised Action Ranking","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11151","citing_title":"RankQ: Offline-to-Online Reinforcement Learning via Self-Supervised Action Ranking","ref_index":21,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/HLBZF6WZC3QCI6XN2TQ4SEFPST","json":"https://pith.science/pith/HLBZF6WZC3QCI6XN2TQ4SEFPST.json","graph_json":"https://pith.science/api/pith-number/HLBZF6WZC3QCI6XN2TQ4SEFPST/graph.json","events_json":"https://pith.science/api/pith-number/HLBZF6WZC3QCI6XN2TQ4SEFPST/events.json","paper":"https://pith.science/paper/HLBZF6WZ"},"agent_actions":{"view_html":"https://pith.science/pith/HLBZF6WZC3QCI6XN2TQ4SEFPST","download_json":"https://pith.science/pith/HLBZF6WZC3QCI6XN2TQ4SEFPST.json","view_paper":"https://pith.science/paper/HLBZF6WZ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2110.01548&json=true","fetch_graph":"https://pith.science/api/pith-number/HLBZF6WZC3QCI6XN2TQ4SEFPST/graph.json","fetch_events":"https://pith.science/api/pith-number/HLBZF6WZC3QCI6XN2TQ4SEFPST/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/HLBZF6WZC3QCI6XN2TQ4SEFPST/action/timestamp_anchor","attest_storage":"https://pith.science/pith/HLBZF6WZC3QCI6XN2TQ4SEFPST/action/storage_attestation","attest_author":"https://pith.science/pith/HLBZF6WZC3QCI6XN2TQ4SEFPST/action/author_attestation","sign_citation":"https://pith.science/pith/HLBZF6WZC3QCI6XN2TQ4SEFPST/action/citation_signature","submit_replication":"https://pith.science/pith/HLBZF6WZC3QCI6XN2TQ4SEFPST/action/replication_record"}},"created_at":"2026-07-05T03:20:04.260260+00:00","updated_at":"2026-07-05T03:20:04.260260+00:00"}