{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2022:D7ODVK3RASZYIZ6MGMZV52XPAP","short_pith_number":"pith:D7ODVK3R","canonical_record":{"source":{"id":"2206.04745","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2022-06-09T19:44:35Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"4dbfecc601c0c32d5374f95fb862dc80f0b456449d4ddc0ea3bedc036e2bfecf","abstract_canon_sha256":"677bb6fb327cb7ece27c862a34bcbf2a463804dfe8a6cb539d04c8c4c7850968"},"schema_version":"1.0"},"canonical_sha256":"1fdc3aab7104b38467cc33335eeaef03f3baade5c843cfb50ae9ae38f0f4d16a","source":{"kind":"arxiv","id":"2206.04745","version":3},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2206.04745","created_at":"2026-07-05T07:47:23Z"},{"alias_kind":"arxiv_version","alias_value":"2206.04745v3","created_at":"2026-07-05T07:47:23Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2206.04745","created_at":"2026-07-05T07:47:23Z"},{"alias_kind":"pith_short_12","alias_value":"D7ODVK3RASZY","created_at":"2026-07-05T07:47:23Z"},{"alias_kind":"pith_short_16","alias_value":"D7ODVK3RASZYIZ6M","created_at":"2026-07-05T07:47:23Z"},{"alias_kind":"pith_short_8","alias_value":"D7ODVK3R","created_at":"2026-07-05T07:47:23Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2022:D7ODVK3RASZYIZ6MGMZV52XPAP","target":"record","payload":{"canonical_record":{"source":{"id":"2206.04745","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2022-06-09T19:44:35Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"4dbfecc601c0c32d5374f95fb862dc80f0b456449d4ddc0ea3bedc036e2bfecf","abstract_canon_sha256":"677bb6fb327cb7ece27c862a34bcbf2a463804dfe8a6cb539d04c8c4c7850968"},"schema_version":"1.0"},"canonical_sha256":"1fdc3aab7104b38467cc33335eeaef03f3baade5c843cfb50ae9ae38f0f4d16a","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:47:23.751483Z","signature_b64":"MMiWpvAUbbsHgH6WAxRJMT9mn4gwleU8Kggobnbw856xBfeBvL25S25W64bLxqz04N6pOaNrlZiLWFYS+d7HAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"1fdc3aab7104b38467cc33335eeaef03f3baade5c843cfb50ae9ae38f0f4d16a","last_reissued_at":"2026-07-05T07:47:23.751125Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:47:23.751125Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2206.04745","source_version":3,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T07:47:23Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"zzLEm1rI1m+pbvkQvVlguFLqlwysLjrcH7AxH+9/aEpfAGm2ChGobS/2kmAVugOphXZmppbtC97DFLWcVHlDBg==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-11T16:09:07.798954Z"},"content_sha256":"af697ae0562fadca9cb02e1b71bf5ef2e5bdbe5747a691dc66af071d396c28bb","schema_version":"1.0","event_id":"sha256:af697ae0562fadca9cb02e1b71bf5ef2e5bdbe5747a691dc66af071d396c28bb"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2022:D7ODVK3RASZYIZ6MGMZV52XPAP","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Mildly Conservative Q-Learning for Offline Reinforcement Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Jiafei Lyu, Xiaoteng Ma, Xiu Li, Zongqing Lu","submitted_at":"2022-06-09T19:44:35Z","abstract_excerpt":"Offline reinforcement learning (RL) defines the task of learning from a static logged dataset without continually interacting with the environment. The distribution shift between the learned policy and the behavior policy makes it necessary for the value function to stay conservative such that out-of-distribution (OOD) actions will not be severely overestimated. However, existing approaches, penalizing the unseen actions or regularizing with the behavior policy, are too pessimistic, which suppresses the generalization of the value function and hinders the performance improvement. This paper ex"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2206.04745","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2206.04745/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T07:47:23Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"gbVdBQhflRvuGPHIcauQGrD2pe6PByPilVPu6f3VMLE/YeYTv1hVPu2w8++5WQ/4M2+go/DfaYh6KplOmjc8CA==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-11T16:09:07.799322Z"},"content_sha256":"8d04b9238e72e357953134c441e1cf6db8aa5813e7d4189b05d35035c5dff988","schema_version":"1.0","event_id":"sha256:8d04b9238e72e357953134c441e1cf6db8aa5813e7d4189b05d35035c5dff988"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/D7ODVK3RASZYIZ6MGMZV52XPAP/bundle.json","state_url":"https://pith.science/pith/D7ODVK3RASZYIZ6MGMZV52XPAP/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/D7ODVK3RASZYIZ6MGMZV52XPAP/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-11T16:09:07Z","links":{"resolver":"https://pith.science/pith/D7ODVK3RASZYIZ6MGMZV52XPAP","bundle":"https://pith.science/pith/D7ODVK3RASZYIZ6MGMZV52XPAP/bundle.json","state":"https://pith.science/pith/D7ODVK3RASZYIZ6MGMZV52XPAP/state.json","well_known_bundle":"https://pith.science/.well-known/pith/D7ODVK3RASZYIZ6MGMZV52XPAP/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2022:D7ODVK3RASZYIZ6MGMZV52XPAP","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"677bb6fb327cb7ece27c862a34bcbf2a463804dfe8a6cb539d04c8c4c7850968","cross_cats_sorted":["cs.AI"],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2022-06-09T19:44:35Z","title_canon_sha256":"4dbfecc601c0c32d5374f95fb862dc80f0b456449d4ddc0ea3bedc036e2bfecf"},"schema_version":"1.0","source":{"id":"2206.04745","kind":"arxiv","version":3}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2206.04745","created_at":"2026-07-05T07:47:23Z"},{"alias_kind":"arxiv_version","alias_value":"2206.04745v3","created_at":"2026-07-05T07:47:23Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2206.04745","created_at":"2026-07-05T07:47:23Z"},{"alias_kind":"pith_short_12","alias_value":"D7ODVK3RASZY","created_at":"2026-07-05T07:47:23Z"},{"alias_kind":"pith_short_16","alias_value":"D7ODVK3RASZYIZ6M","created_at":"2026-07-05T07:47:23Z"},{"alias_kind":"pith_short_8","alias_value":"D7ODVK3R","created_at":"2026-07-05T07:47:23Z"}],"graph_snapshots":[{"event_id":"sha256:8d04b9238e72e357953134c441e1cf6db8aa5813e7d4189b05d35035c5dff988","target":"graph","created_at":"2026-07-05T07:47:23Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2206.04745/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Offline reinforcement learning (RL) defines the task of learning from a static logged dataset without continually interacting with the environment. The distribution shift between the learned policy and the behavior policy makes it necessary for the value function to stay conservative such that out-of-distribution (OOD) actions will not be severely overestimated. However, existing approaches, penalizing the unseen actions or regularizing with the behavior policy, are too pessimistic, which suppresses the generalization of the value function and hinders the performance improvement. This paper ex","authors_text":"Jiafei Lyu, Xiaoteng Ma, Xiu Li, Zongqing Lu","cross_cats":["cs.AI"],"headline":"","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2022-06-09T19:44:35Z","title":"Mildly Conservative Q-Learning for Offline Reinforcement Learning"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2206.04745","kind":"arxiv","version":3},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:af697ae0562fadca9cb02e1b71bf5ef2e5bdbe5747a691dc66af071d396c28bb","target":"record","created_at":"2026-07-05T07:47:23Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"677bb6fb327cb7ece27c862a34bcbf2a463804dfe8a6cb539d04c8c4c7850968","cross_cats_sorted":["cs.AI"],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2022-06-09T19:44:35Z","title_canon_sha256":"4dbfecc601c0c32d5374f95fb862dc80f0b456449d4ddc0ea3bedc036e2bfecf"},"schema_version":"1.0","source":{"id":"2206.04745","kind":"arxiv","version":3}},"canonical_sha256":"1fdc3aab7104b38467cc33335eeaef03f3baade5c843cfb50ae9ae38f0f4d16a","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"1fdc3aab7104b38467cc33335eeaef03f3baade5c843cfb50ae9ae38f0f4d16a","first_computed_at":"2026-07-05T07:47:23.751125Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T07:47:23.751125Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"MMiWpvAUbbsHgH6WAxRJMT9mn4gwleU8Kggobnbw856xBfeBvL25S25W64bLxqz04N6pOaNrlZiLWFYS+d7HAg==","signature_status":"signed_v1","signed_at":"2026-07-05T07:47:23.751483Z","signed_message":"canonical_sha256_bytes"},"source_id":"2206.04745","source_kind":"arxiv","source_version":3}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:af697ae0562fadca9cb02e1b71bf5ef2e5bdbe5747a691dc66af071d396c28bb","sha256:8d04b9238e72e357953134c441e1cf6db8aa5813e7d4189b05d35035c5dff988"],"state_sha256":"6e665e90cd2933860840c09514709e594f8e83c308cc1afabe9906f7d73e5e59"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"KZ4X9hTtxDZcRL++cGiUA0aA2ijJE1Z+cCKthzTM0PfAXm1LI3Zm4nIr2lqiUaG1R5ZXqd3wNFnF5nqQb77RAA==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-11T16:09:07.801823Z","bundle_sha256":"0a873b1a9753038dde60402a2f9eb98f4ca8ebba727220b634af871659002508"}}