{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:NKBGIK44FXNG4UPQQMOFZY4HED","short_pith_number":"pith:NKBGIK44","schema_version":"1.0","canonical_sha256":"6a82642b9c2dda6e51f0831c5ce38720c1eff7678a715846ed3925fb710f00c7","source":{"kind":"arxiv","id":"2309.06835","version":1},"attestation_state":"computed","paper":{"title":"Safe Reinforcement Learning with Dual Robustness","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Chuxiong Hu, Shengbo Eben Li, Yujie Yang, Yunan Wang, Zeyang Li","submitted_at":"2023-09-13T09:34:21Z","abstract_excerpt":"Reinforcement learning (RL) agents are vulnerable to adversarial disturbances, which can deteriorate task performance or compromise safety specifications. Existing methods either address safety requirements under the assumption of no adversary (e.g., safe RL) or only focus on robustness against performance adversaries (e.g., robust RL). Learning one policy that is both safe and robust remains a challenging open problem. The difficulty is how to tackle two intertwined aspects in the worst cases: feasibility and optimality. Optimality is only valid inside a feasible region, while identification "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2309.06835","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2023-09-13T09:34:21Z","cross_cats_sorted":[],"title_canon_sha256":"7643c09687c6c7e319b7119127f73dea92b39f29b967d6b792609903d8197aaa","abstract_canon_sha256":"9fdb829106a0523e5f44c161d8e26194c781d09db9c1c86380f9d208f3a8b74a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:50:23.220199Z","signature_b64":"yZXyQo5eSdhxz9eXy4eLL4Ru7fOUNYXZ2y4IngeXG4f5APzSM/70SbC5M0d2ovXYw7MK35D5ui4UOCcSQIYPAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"6a82642b9c2dda6e51f0831c5ce38720c1eff7678a715846ed3925fb710f00c7","last_reissued_at":"2026-07-05T06:50:23.219745Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:50:23.219745Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Safe Reinforcement Learning with Dual Robustness","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Chuxiong Hu, Shengbo Eben Li, Yujie Yang, Yunan Wang, Zeyang Li","submitted_at":"2023-09-13T09:34:21Z","abstract_excerpt":"Reinforcement learning (RL) agents are vulnerable to adversarial disturbances, which can deteriorate task performance or compromise safety specifications. Existing methods either address safety requirements under the assumption of no adversary (e.g., safe RL) or only focus on robustness against performance adversaries (e.g., robust RL). Learning one policy that is both safe and robust remains a challenging open problem. The difficulty is how to tackle two intertwined aspects in the worst cases: feasibility and optimality. Optimality is only valid inside a feasible region, while identification "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2309.06835","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2309.06835/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2309.06835","created_at":"2026-07-05T06:50:23.219799+00:00"},{"alias_kind":"arxiv_version","alias_value":"2309.06835v1","created_at":"2026-07-05T06:50:23.219799+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2309.06835","created_at":"2026-07-05T06:50:23.219799+00:00"},{"alias_kind":"pith_short_12","alias_value":"NKBGIK44FXNG","created_at":"2026-07-05T06:50:23.219799+00:00"},{"alias_kind":"pith_short_16","alias_value":"NKBGIK44FXNG4UPQ","created_at":"2026-07-05T06:50:23.219799+00:00"},{"alias_kind":"pith_short_8","alias_value":"NKBGIK44","created_at":"2026-07-05T06:50:23.219799+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2502.09417","citing_title":"A Survey of Reinforcement Learning for Optimization in Automation","ref_index":80,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/NKBGIK44FXNG4UPQQMOFZY4HED","json":"https://pith.science/pith/NKBGIK44FXNG4UPQQMOFZY4HED.json","graph_json":"https://pith.science/api/pith-number/NKBGIK44FXNG4UPQQMOFZY4HED/graph.json","events_json":"https://pith.science/api/pith-number/NKBGIK44FXNG4UPQQMOFZY4HED/events.json","paper":"https://pith.science/paper/NKBGIK44"},"agent_actions":{"view_html":"https://pith.science/pith/NKBGIK44FXNG4UPQQMOFZY4HED","download_json":"https://pith.science/pith/NKBGIK44FXNG4UPQQMOFZY4HED.json","view_paper":"https://pith.science/paper/NKBGIK44","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2309.06835&json=true","fetch_graph":"https://pith.science/api/pith-number/NKBGIK44FXNG4UPQQMOFZY4HED/graph.json","fetch_events":"https://pith.science/api/pith-number/NKBGIK44FXNG4UPQQMOFZY4HED/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/NKBGIK44FXNG4UPQQMOFZY4HED/action/timestamp_anchor","attest_storage":"https://pith.science/pith/NKBGIK44FXNG4UPQQMOFZY4HED/action/storage_attestation","attest_author":"https://pith.science/pith/NKBGIK44FXNG4UPQQMOFZY4HED/action/author_attestation","sign_citation":"https://pith.science/pith/NKBGIK44FXNG4UPQQMOFZY4HED/action/citation_signature","submit_replication":"https://pith.science/pith/NKBGIK44FXNG4UPQQMOFZY4HED/action/replication_record"}},"created_at":"2026-07-05T06:50:23.219799+00:00","updated_at":"2026-07-05T06:50:23.219799+00:00"}