{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2026:6X7HSZ56FX52VZPOB3T3KE3ULP","short_pith_number":"pith:6X7HSZ56","canonical_record":{"source":{"id":"2607.19199","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2026-07-21T15:34:06Z","cross_cats_sorted":[],"title_canon_sha256":"fb4e9c0655721ef9fd1ab30939b3cb95dc4dbc435f1a40adae705c40d211d616","abstract_canon_sha256":"811797e2015de9d3034216d1bfac69e763ed45d3806e3bf05700a4ebc2e18913"},"schema_version":"1.0"},"canonical_sha256":"f5fe7967be2dfbaae5ee0ee7b513745bfde42dbf266610e8cfa10175c5cd26ea","source":{"kind":"arxiv","id":"2607.19199","version":1},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2607.19199","created_at":"2026-07-22T01:24:13Z"},{"alias_kind":"arxiv_version","alias_value":"2607.19199v1","created_at":"2026-07-22T01:24:13Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2607.19199","created_at":"2026-07-22T01:24:13Z"},{"alias_kind":"pith_short_12","alias_value":"6X7HSZ56FX52","created_at":"2026-07-22T01:24:13Z"},{"alias_kind":"pith_short_16","alias_value":"6X7HSZ56FX52VZPO","created_at":"2026-07-22T01:24:13Z"},{"alias_kind":"pith_short_8","alias_value":"6X7HSZ56","created_at":"2026-07-22T01:24:13Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2026:6X7HSZ56FX52VZPOB3T3KE3ULP","target":"record","payload":{"canonical_record":{"source":{"id":"2607.19199","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2026-07-21T15:34:06Z","cross_cats_sorted":[],"title_canon_sha256":"fb4e9c0655721ef9fd1ab30939b3cb95dc4dbc435f1a40adae705c40d211d616","abstract_canon_sha256":"811797e2015de9d3034216d1bfac69e763ed45d3806e3bf05700a4ebc2e18913"},"schema_version":"1.0"},"canonical_sha256":"f5fe7967be2dfbaae5ee0ee7b513745bfde42dbf266610e8cfa10175c5cd26ea","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-22T01:24:13.840612Z","signature_b64":"5jx29zvojpt+Uphboixji3/c0cZAd9wLo/TduaO6Jo9FZbDuL/wGbEXNYlxlGibxnrxvxkTl6j0gBJbD3ySQBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f5fe7967be2dfbaae5ee0ee7b513745bfde42dbf266610e8cfa10175c5cd26ea","last_reissued_at":"2026-07-22T01:24:13.839709Z","signature_status":"signed_v1","first_computed_at":"2026-07-22T01:24:13.839709Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2607.19199","source_version":1,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-22T01:24:13Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"XQu0YLHCc0yX/eNFfL0jubbAuH2T01PmKqBSchDKLOpvIBfDUJCu7zrnqHXHdQcMP0phB6twFbF0e7wZSWlKCw==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-04T07:13:10.738060Z"},"content_sha256":"4e40366897e7b27274627eb26bbc68ae7b9c9ab16cdf5adbd6b4b319bee5d3f2","schema_version":"1.0","event_id":"sha256:4e40366897e7b27274627eb26bbc68ae7b9c9ab16cdf5adbd6b4b319bee5d3f2"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2026:6X7HSZ56FX52VZPOB3T3KE3ULP","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Conservative Query and Adaptive Regularization for Offline RL Under Uncertainty Estimation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Li-Rong Zhou, Qin-Wen Luo, Sheng-Jun Huang","submitted_at":"2026-07-21T15:34:06Z","abstract_excerpt":"Offline reinforcement learning (RL) aims to learn an effective policy from a static dataset, but its performance is fundamentally limited by dataset coverage. Action preference queries leverage expert feedback without additional environment interaction, enabling policy improvement during offline training. However, existing methods still face two key challenges: selecting informative preference queries and effectively exploiting the collected feedback. Current approaches typically rely only on the distance between policy actions and dataset actions for query selection, while enforcing fixed con"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2607.19199","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2607.19199/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-22T01:24:13Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"7edCGPGLyFJ9N271LHQfOvDQbhvrFsJyS1iYmt/sFfzOWN7iaQcGo+L2j9v6oXNC9C3N98WMrpXjXrS56k3pAg==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-04T07:13:10.738587Z"},"content_sha256":"302685380cc212093f9b72bd57341d984e17111daba84d864c703ad85d04a8db","schema_version":"1.0","event_id":"sha256:302685380cc212093f9b72bd57341d984e17111daba84d864c703ad85d04a8db"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/6X7HSZ56FX52VZPOB3T3KE3ULP/bundle.json","state_url":"https://pith.science/pith/6X7HSZ56FX52VZPOB3T3KE3ULP/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/6X7HSZ56FX52VZPOB3T3KE3ULP/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-04T07:13:10Z","links":{"resolver":"https://pith.science/pith/6X7HSZ56FX52VZPOB3T3KE3ULP","bundle":"https://pith.science/pith/6X7HSZ56FX52VZPOB3T3KE3ULP/bundle.json","state":"https://pith.science/pith/6X7HSZ56FX52VZPOB3T3KE3ULP/state.json","well_known_bundle":"https://pith.science/.well-known/pith/6X7HSZ56FX52VZPOB3T3KE3ULP/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2026:6X7HSZ56FX52VZPOB3T3KE3ULP","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"811797e2015de9d3034216d1bfac69e763ed45d3806e3bf05700a4ebc2e18913","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2026-07-21T15:34:06Z","title_canon_sha256":"fb4e9c0655721ef9fd1ab30939b3cb95dc4dbc435f1a40adae705c40d211d616"},"schema_version":"1.0","source":{"id":"2607.19199","kind":"arxiv","version":1}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2607.19199","created_at":"2026-07-22T01:24:13Z"},{"alias_kind":"arxiv_version","alias_value":"2607.19199v1","created_at":"2026-07-22T01:24:13Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2607.19199","created_at":"2026-07-22T01:24:13Z"},{"alias_kind":"pith_short_12","alias_value":"6X7HSZ56FX52","created_at":"2026-07-22T01:24:13Z"},{"alias_kind":"pith_short_16","alias_value":"6X7HSZ56FX52VZPO","created_at":"2026-07-22T01:24:13Z"},{"alias_kind":"pith_short_8","alias_value":"6X7HSZ56","created_at":"2026-07-22T01:24:13Z"}],"graph_snapshots":[{"event_id":"sha256:302685380cc212093f9b72bd57341d984e17111daba84d864c703ad85d04a8db","target":"graph","created_at":"2026-07-22T01:24:13Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2607.19199/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Offline reinforcement learning (RL) aims to learn an effective policy from a static dataset, but its performance is fundamentally limited by dataset coverage. Action preference queries leverage expert feedback without additional environment interaction, enabling policy improvement during offline training. However, existing methods still face two key challenges: selecting informative preference queries and effectively exploiting the collected feedback. Current approaches typically rely only on the distance between policy actions and dataset actions for query selection, while enforcing fixed con","authors_text":"Li-Rong Zhou, Qin-Wen Luo, Sheng-Jun Huang","cross_cats":[],"headline":"","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2026-07-21T15:34:06Z","title":"Conservative Query and Adaptive Regularization for Offline RL Under Uncertainty Estimation"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2607.19199","kind":"arxiv","version":1},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:4e40366897e7b27274627eb26bbc68ae7b9c9ab16cdf5adbd6b4b319bee5d3f2","target":"record","created_at":"2026-07-22T01:24:13Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"811797e2015de9d3034216d1bfac69e763ed45d3806e3bf05700a4ebc2e18913","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2026-07-21T15:34:06Z","title_canon_sha256":"fb4e9c0655721ef9fd1ab30939b3cb95dc4dbc435f1a40adae705c40d211d616"},"schema_version":"1.0","source":{"id":"2607.19199","kind":"arxiv","version":1}},"canonical_sha256":"f5fe7967be2dfbaae5ee0ee7b513745bfde42dbf266610e8cfa10175c5cd26ea","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"f5fe7967be2dfbaae5ee0ee7b513745bfde42dbf266610e8cfa10175c5cd26ea","first_computed_at":"2026-07-22T01:24:13.839709Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-22T01:24:13.839709Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"5jx29zvojpt+Uphboixji3/c0cZAd9wLo/TduaO6Jo9FZbDuL/wGbEXNYlxlGibxnrxvxkTl6j0gBJbD3ySQBA==","signature_status":"signed_v1","signed_at":"2026-07-22T01:24:13.840612Z","signed_message":"canonical_sha256_bytes"},"source_id":"2607.19199","source_kind":"arxiv","source_version":1}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:4e40366897e7b27274627eb26bbc68ae7b9c9ab16cdf5adbd6b4b319bee5d3f2","sha256:302685380cc212093f9b72bd57341d984e17111daba84d864c703ad85d04a8db"],"state_sha256":"a692c3ccd82e0be48038f7ba309a7e588b374f2cc3bdf34171864ecf921d3a3e"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"4+xheve0BEfmvWzRfTJQOo7j+yJXgfHapJF5zedTu/9Rjsor3zdlLcsKL9d0AvjraE9U4pCaI13gr8jj90kqDg==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-04T07:13:10.743991Z","bundle_sha256":"dd57ec8d27da99e22d4cd003759038eef89311c3852cedfa9d969b4ff3847d46"}}