{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2024:E6MQLNG7U2VSL4UQKV3KQ3IZFL","short_pith_number":"pith:E6MQLNG7","canonical_record":{"source":{"id":"2401.14758","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-01-26T10:33:38Z","cross_cats_sorted":[],"title_canon_sha256":"6255c33be9a394a7ed17dcc44ed604814039c7b88eea419888a1811ce7f0feed","abstract_canon_sha256":"797a7433ed33f644ed15fb619922fa30db2628765c9d9c3d8e82e0d354fce476"},"schema_version":"1.0"},"canonical_sha256":"279905b4dfa6ab25f2905576a86d192ae28e63c23e582bade4d949fb18375afc","source":{"kind":"arxiv","id":"2401.14758","version":2},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2401.14758","created_at":"2026-07-05T08:08:01Z"},{"alias_kind":"arxiv_version","alias_value":"2401.14758v2","created_at":"2026-07-05T08:08:01Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2401.14758","created_at":"2026-07-05T08:08:01Z"},{"alias_kind":"pith_short_12","alias_value":"E6MQLNG7U2VS","created_at":"2026-07-05T08:08:01Z"},{"alias_kind":"pith_short_16","alias_value":"E6MQLNG7U2VSL4UQ","created_at":"2026-07-05T08:08:01Z"},{"alias_kind":"pith_short_8","alias_value":"E6MQLNG7","created_at":"2026-07-05T08:08:01Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2024:E6MQLNG7U2VSL4UQKV3KQ3IZFL","target":"record","payload":{"canonical_record":{"source":{"id":"2401.14758","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-01-26T10:33:38Z","cross_cats_sorted":[],"title_canon_sha256":"6255c33be9a394a7ed17dcc44ed604814039c7b88eea419888a1811ce7f0feed","abstract_canon_sha256":"797a7433ed33f644ed15fb619922fa30db2628765c9d9c3d8e82e0d354fce476"},"schema_version":"1.0"},"canonical_sha256":"279905b4dfa6ab25f2905576a86d192ae28e63c23e582bade4d949fb18375afc","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:08:01.465035Z","signature_b64":"k6tqc6d/8nGpam3tLqBBC+8d3EhjFrA3uPaTlNkKtDgpYnHU/rdkphFTECfPUUNShKZLNfvr9e3GpmZ3yMj4Cw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"279905b4dfa6ab25f2905576a86d192ae28e63c23e582bade4d949fb18375afc","last_reissued_at":"2026-07-05T08:08:01.464449Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:08:01.464449Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2401.14758","source_version":2,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T08:08:01Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"7Yh6Yj8lOmLWYEkueRzebUrP6MoT/ZobRBOBdH4vRmqV/K/3WZYQPjpnv0eTVETGC6o/EM2/S6Lynjd8zJUcDg==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-03T19:42:16.759790Z"},"content_sha256":"906926cf9f8826ae0abe47f785da227d51e2e6f4acf28ff31beba067432ef219","schema_version":"1.0","event_id":"sha256:906926cf9f8826ae0abe47f785da227d51e2e6f4acf28ff31beba067432ef219"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2024:E6MQLNG7U2VSL4UQKV3KQ3IZFL","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Off-Policy Primal-Dual Safe Reinforcement Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Bo Tang, Chao Yu, Dong Wang, Qian Lin, Qianlong Xie, Shangqin Mao, Xingxing Wang, Zifan Wu","submitted_at":"2024-01-26T10:33:38Z","abstract_excerpt":"Primal-dual safe RL methods commonly perform iterations between the primal update of the policy and the dual update of the Lagrange Multiplier. Such a training paradigm is highly susceptible to the error in cumulative cost estimation since this estimation serves as the key bond connecting the primal and dual update processes. We show that this problem causes significant underestimation of cost when using off-policy methods, leading to the failure to satisfy the safety constraint. To address this issue, we propose conservative policy optimization, which learns a policy in a constraint-satisfyin"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2401.14758","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2401.14758/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T08:08:01Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"ZpnB97qqwGcVb9BgUsTL22tmtj1OxVXogD6SDP8Yvp5awpcV3NAy4xJEa+aerN6kwj5wEGR8KxxE4Old38zmAg==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-03T19:42:16.763388Z"},"content_sha256":"22c8e5108e16161aa9710f01c073c7a5aa619d8b5149e0cefdc236674a7b855a","schema_version":"1.0","event_id":"sha256:22c8e5108e16161aa9710f01c073c7a5aa619d8b5149e0cefdc236674a7b855a"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/E6MQLNG7U2VSL4UQKV3KQ3IZFL/bundle.json","state_url":"https://pith.science/pith/E6MQLNG7U2VSL4UQKV3KQ3IZFL/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/E6MQLNG7U2VSL4UQKV3KQ3IZFL/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-03T19:42:16Z","links":{"resolver":"https://pith.science/pith/E6MQLNG7U2VSL4UQKV3KQ3IZFL","bundle":"https://pith.science/pith/E6MQLNG7U2VSL4UQKV3KQ3IZFL/bundle.json","state":"https://pith.science/pith/E6MQLNG7U2VSL4UQKV3KQ3IZFL/state.json","well_known_bundle":"https://pith.science/.well-known/pith/E6MQLNG7U2VSL4UQKV3KQ3IZFL/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2024:E6MQLNG7U2VSL4UQKV3KQ3IZFL","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"797a7433ed33f644ed15fb619922fa30db2628765c9d9c3d8e82e0d354fce476","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-01-26T10:33:38Z","title_canon_sha256":"6255c33be9a394a7ed17dcc44ed604814039c7b88eea419888a1811ce7f0feed"},"schema_version":"1.0","source":{"id":"2401.14758","kind":"arxiv","version":2}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2401.14758","created_at":"2026-07-05T08:08:01Z"},{"alias_kind":"arxiv_version","alias_value":"2401.14758v2","created_at":"2026-07-05T08:08:01Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2401.14758","created_at":"2026-07-05T08:08:01Z"},{"alias_kind":"pith_short_12","alias_value":"E6MQLNG7U2VS","created_at":"2026-07-05T08:08:01Z"},{"alias_kind":"pith_short_16","alias_value":"E6MQLNG7U2VSL4UQ","created_at":"2026-07-05T08:08:01Z"},{"alias_kind":"pith_short_8","alias_value":"E6MQLNG7","created_at":"2026-07-05T08:08:01Z"}],"graph_snapshots":[{"event_id":"sha256:22c8e5108e16161aa9710f01c073c7a5aa619d8b5149e0cefdc236674a7b855a","target":"graph","created_at":"2026-07-05T08:08:01Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2401.14758/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Primal-dual safe RL methods commonly perform iterations between the primal update of the policy and the dual update of the Lagrange Multiplier. Such a training paradigm is highly susceptible to the error in cumulative cost estimation since this estimation serves as the key bond connecting the primal and dual update processes. We show that this problem causes significant underestimation of cost when using off-policy methods, leading to the failure to satisfy the safety constraint. To address this issue, we propose conservative policy optimization, which learns a policy in a constraint-satisfyin","authors_text":"Bo Tang, Chao Yu, Dong Wang, Qian Lin, Qianlong Xie, Shangqin Mao, Xingxing Wang, Zifan Wu","cross_cats":[],"headline":"","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-01-26T10:33:38Z","title":"Off-Policy Primal-Dual Safe Reinforcement Learning"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2401.14758","kind":"arxiv","version":2},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:906926cf9f8826ae0abe47f785da227d51e2e6f4acf28ff31beba067432ef219","target":"record","created_at":"2026-07-05T08:08:01Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"797a7433ed33f644ed15fb619922fa30db2628765c9d9c3d8e82e0d354fce476","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-01-26T10:33:38Z","title_canon_sha256":"6255c33be9a394a7ed17dcc44ed604814039c7b88eea419888a1811ce7f0feed"},"schema_version":"1.0","source":{"id":"2401.14758","kind":"arxiv","version":2}},"canonical_sha256":"279905b4dfa6ab25f2905576a86d192ae28e63c23e582bade4d949fb18375afc","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"279905b4dfa6ab25f2905576a86d192ae28e63c23e582bade4d949fb18375afc","first_computed_at":"2026-07-05T08:08:01.464449Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T08:08:01.464449Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"k6tqc6d/8nGpam3tLqBBC+8d3EhjFrA3uPaTlNkKtDgpYnHU/rdkphFTECfPUUNShKZLNfvr9e3GpmZ3yMj4Cw==","signature_status":"signed_v1","signed_at":"2026-07-05T08:08:01.465035Z","signed_message":"canonical_sha256_bytes"},"source_id":"2401.14758","source_kind":"arxiv","source_version":2}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:906926cf9f8826ae0abe47f785da227d51e2e6f4acf28ff31beba067432ef219","sha256:22c8e5108e16161aa9710f01c073c7a5aa619d8b5149e0cefdc236674a7b855a"],"state_sha256":"16569509c468a79a8c9c6aa723459cdcd6432f3d65afaf55d7350d1035f6c998"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"I/s915lVDHtX3p2alF+MBnj+VPY8tUKXtB8IUpck+/n2iS3Z9TyK/BApJRQ+1F/7zXQX4tJa5dH0rCjmVsdEAA==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-03T19:42:16.782058Z","bundle_sha256":"0507965505ee5bf61b4a5f9bc787d1b88e6f25e71b96a1552d046ab4ed28e5b5"}}