{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2023:NKBGIK44FXNG4UPQQMOFZY4HED","short_pith_number":"pith:NKBGIK44","canonical_record":{"source":{"id":"2309.06835","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2023-09-13T09:34:21Z","cross_cats_sorted":[],"title_canon_sha256":"7643c09687c6c7e319b7119127f73dea92b39f29b967d6b792609903d8197aaa","abstract_canon_sha256":"9fdb829106a0523e5f44c161d8e26194c781d09db9c1c86380f9d208f3a8b74a"},"schema_version":"1.0"},"canonical_sha256":"6a82642b9c2dda6e51f0831c5ce38720c1eff7678a715846ed3925fb710f00c7","source":{"kind":"arxiv","id":"2309.06835","version":1},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2309.06835","created_at":"2026-07-05T06:50:23Z"},{"alias_kind":"arxiv_version","alias_value":"2309.06835v1","created_at":"2026-07-05T06:50:23Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2309.06835","created_at":"2026-07-05T06:50:23Z"},{"alias_kind":"pith_short_12","alias_value":"NKBGIK44FXNG","created_at":"2026-07-05T06:50:23Z"},{"alias_kind":"pith_short_16","alias_value":"NKBGIK44FXNG4UPQ","created_at":"2026-07-05T06:50:23Z"},{"alias_kind":"pith_short_8","alias_value":"NKBGIK44","created_at":"2026-07-05T06:50:23Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2023:NKBGIK44FXNG4UPQQMOFZY4HED","target":"record","payload":{"canonical_record":{"source":{"id":"2309.06835","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2023-09-13T09:34:21Z","cross_cats_sorted":[],"title_canon_sha256":"7643c09687c6c7e319b7119127f73dea92b39f29b967d6b792609903d8197aaa","abstract_canon_sha256":"9fdb829106a0523e5f44c161d8e26194c781d09db9c1c86380f9d208f3a8b74a"},"schema_version":"1.0"},"canonical_sha256":"6a82642b9c2dda6e51f0831c5ce38720c1eff7678a715846ed3925fb710f00c7","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:50:23.220199Z","signature_b64":"yZXyQo5eSdhxz9eXy4eLL4Ru7fOUNYXZ2y4IngeXG4f5APzSM/70SbC5M0d2ovXYw7MK35D5ui4UOCcSQIYPAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"6a82642b9c2dda6e51f0831c5ce38720c1eff7678a715846ed3925fb710f00c7","last_reissued_at":"2026-07-05T06:50:23.219745Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:50:23.219745Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2309.06835","source_version":1,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T06:50:23Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"M0kYc/h3e727iANUMIG18esFo9qqgPSFwTpmbIgb8XbOiCbjWYuA+plFNy6+yE+ETIkH+lvxtbxklPVIVK4cCQ==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-08T14:15:16.647893Z"},"content_sha256":"4ad5c9b8c850bb8f71891b094e624e29bacd2b780e46308f158d98f97ed1c6fc","schema_version":"1.0","event_id":"sha256:4ad5c9b8c850bb8f71891b094e624e29bacd2b780e46308f158d98f97ed1c6fc"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2023:NKBGIK44FXNG4UPQQMOFZY4HED","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Safe Reinforcement Learning with Dual Robustness","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Chuxiong Hu, Shengbo Eben Li, Yujie Yang, Yunan Wang, Zeyang Li","submitted_at":"2023-09-13T09:34:21Z","abstract_excerpt":"Reinforcement learning (RL) agents are vulnerable to adversarial disturbances, which can deteriorate task performance or compromise safety specifications. Existing methods either address safety requirements under the assumption of no adversary (e.g., safe RL) or only focus on robustness against performance adversaries (e.g., robust RL). Learning one policy that is both safe and robust remains a challenging open problem. The difficulty is how to tackle two intertwined aspects in the worst cases: feasibility and optimality. Optimality is only valid inside a feasible region, while identification "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2309.06835","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2309.06835/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T06:50:23Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"hn59bzqjKNI75JJWPPe1onfL6i8d+wfbuoDDukEByfIyroZekA0gGBH5zYtJASnj5KZHgZhu3O3vQVZtvbBABw==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-08T14:15:16.648466Z"},"content_sha256":"2fd43d6986e0815ff2d909d369ea54816d067c9b13e70f9867b014969e1e7f59","schema_version":"1.0","event_id":"sha256:2fd43d6986e0815ff2d909d369ea54816d067c9b13e70f9867b014969e1e7f59"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/NKBGIK44FXNG4UPQQMOFZY4HED/bundle.json","state_url":"https://pith.science/pith/NKBGIK44FXNG4UPQQMOFZY4HED/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/NKBGIK44FXNG4UPQQMOFZY4HED/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-08T14:15:16Z","links":{"resolver":"https://pith.science/pith/NKBGIK44FXNG4UPQQMOFZY4HED","bundle":"https://pith.science/pith/NKBGIK44FXNG4UPQQMOFZY4HED/bundle.json","state":"https://pith.science/pith/NKBGIK44FXNG4UPQQMOFZY4HED/state.json","well_known_bundle":"https://pith.science/.well-known/pith/NKBGIK44FXNG4UPQQMOFZY4HED/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2023:NKBGIK44FXNG4UPQQMOFZY4HED","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"9fdb829106a0523e5f44c161d8e26194c781d09db9c1c86380f9d208f3a8b74a","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2023-09-13T09:34:21Z","title_canon_sha256":"7643c09687c6c7e319b7119127f73dea92b39f29b967d6b792609903d8197aaa"},"schema_version":"1.0","source":{"id":"2309.06835","kind":"arxiv","version":1}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2309.06835","created_at":"2026-07-05T06:50:23Z"},{"alias_kind":"arxiv_version","alias_value":"2309.06835v1","created_at":"2026-07-05T06:50:23Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2309.06835","created_at":"2026-07-05T06:50:23Z"},{"alias_kind":"pith_short_12","alias_value":"NKBGIK44FXNG","created_at":"2026-07-05T06:50:23Z"},{"alias_kind":"pith_short_16","alias_value":"NKBGIK44FXNG4UPQ","created_at":"2026-07-05T06:50:23Z"},{"alias_kind":"pith_short_8","alias_value":"NKBGIK44","created_at":"2026-07-05T06:50:23Z"}],"graph_snapshots":[{"event_id":"sha256:2fd43d6986e0815ff2d909d369ea54816d067c9b13e70f9867b014969e1e7f59","target":"graph","created_at":"2026-07-05T06:50:23Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2309.06835/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Reinforcement learning (RL) agents are vulnerable to adversarial disturbances, which can deteriorate task performance or compromise safety specifications. Existing methods either address safety requirements under the assumption of no adversary (e.g., safe RL) or only focus on robustness against performance adversaries (e.g., robust RL). Learning one policy that is both safe and robust remains a challenging open problem. The difficulty is how to tackle two intertwined aspects in the worst cases: feasibility and optimality. Optimality is only valid inside a feasible region, while identification ","authors_text":"Chuxiong Hu, Shengbo Eben Li, Yujie Yang, Yunan Wang, Zeyang Li","cross_cats":[],"headline":"","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2023-09-13T09:34:21Z","title":"Safe Reinforcement Learning with Dual Robustness"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2309.06835","kind":"arxiv","version":1},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:4ad5c9b8c850bb8f71891b094e624e29bacd2b780e46308f158d98f97ed1c6fc","target":"record","created_at":"2026-07-05T06:50:23Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"9fdb829106a0523e5f44c161d8e26194c781d09db9c1c86380f9d208f3a8b74a","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2023-09-13T09:34:21Z","title_canon_sha256":"7643c09687c6c7e319b7119127f73dea92b39f29b967d6b792609903d8197aaa"},"schema_version":"1.0","source":{"id":"2309.06835","kind":"arxiv","version":1}},"canonical_sha256":"6a82642b9c2dda6e51f0831c5ce38720c1eff7678a715846ed3925fb710f00c7","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"6a82642b9c2dda6e51f0831c5ce38720c1eff7678a715846ed3925fb710f00c7","first_computed_at":"2026-07-05T06:50:23.219745Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T06:50:23.219745Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"yZXyQo5eSdhxz9eXy4eLL4Ru7fOUNYXZ2y4IngeXG4f5APzSM/70SbC5M0d2ovXYw7MK35D5ui4UOCcSQIYPAQ==","signature_status":"signed_v1","signed_at":"2026-07-05T06:50:23.220199Z","signed_message":"canonical_sha256_bytes"},"source_id":"2309.06835","source_kind":"arxiv","source_version":1}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:4ad5c9b8c850bb8f71891b094e624e29bacd2b780e46308f158d98f97ed1c6fc","sha256:2fd43d6986e0815ff2d909d369ea54816d067c9b13e70f9867b014969e1e7f59"],"state_sha256":"78adbb8f1de070b00b15cb592d9809a49ccda027f572f2a219f3e6da55d8126e"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"hj41dFUopsuyqXVQjU8/D2ygYN9zj5HMWgaUEJILACof6IPbyUlV/LML3p/100mn3aY3mjB6yO1wgofl/qamBA==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-08T14:15:16.653762Z","bundle_sha256":"492059b4db7d23e9c4203eaab8d188915b3b106870e31a746595484931031502"}}