{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:WLNNFXWWCEJENOUTOYJJBITJP6","short_pith_number":"pith:WLNNFXWW","schema_version":"1.0","canonical_sha256":"b2dad2ded6111246ba93761290a2697fa4f8412a3f5f06874011a81e88e8fa3b","source":{"kind":"arxiv","id":"2209.08025","version":1},"attestation_state":"computed","paper":{"title":"Trustworthy Reinforcement Learning Against Intrinsic Vulnerabilities: Robustness, Safety, and Generalizability","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Bo Li, Ding Zhao, Mengdi Xu, Peide Huang, Wenhao Ding, Zhepeng Cen, Zuxin Liu","submitted_at":"2022-09-16T16:10:08Z","abstract_excerpt":"A trustworthy reinforcement learning algorithm should be competent in solving challenging real-world problems, including {robustly} handling uncertainties, satisfying {safety} constraints to avoid catastrophic failures, and {generalizing} to unseen scenarios during deployments. This study aims to overview these main perspectives of trustworthy reinforcement learning considering its intrinsic vulnerabilities on robustness, safety, and generalizability. In particular, we give rigorous formulations, categorize corresponding methodologies, and discuss benchmarks for each perspective. Moreover, we "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2209.08025","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2022-09-16T16:10:08Z","cross_cats_sorted":[],"title_canon_sha256":"24210ae9f49bfe40cd7126c2ed5b0fe05081bcf48308320d9e679b26e354eaf2","abstract_canon_sha256":"b8d2b491144262f9749431c14006e6a381b9f81a3efafcd5624080f52aa9104f"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T04:58:12.808725Z","signature_b64":"/RrHIsCstsyFQzTWcflYdYrOnLV6hXTLjOVd6vPbkszoCxYuvxU0kRon3xWgVUJWqWWHq6QpqkR+dYInDlUpBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b2dad2ded6111246ba93761290a2697fa4f8412a3f5f06874011a81e88e8fa3b","last_reissued_at":"2026-07-05T04:58:12.808252Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T04:58:12.808252Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Trustworthy Reinforcement Learning Against Intrinsic Vulnerabilities: Robustness, Safety, and Generalizability","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Bo Li, Ding Zhao, Mengdi Xu, Peide Huang, Wenhao Ding, Zhepeng Cen, Zuxin Liu","submitted_at":"2022-09-16T16:10:08Z","abstract_excerpt":"A trustworthy reinforcement learning algorithm should be competent in solving challenging real-world problems, including {robustly} handling uncertainties, satisfying {safety} constraints to avoid catastrophic failures, and {generalizing} to unseen scenarios during deployments. This study aims to overview these main perspectives of trustworthy reinforcement learning considering its intrinsic vulnerabilities on robustness, safety, and generalizability. In particular, we give rigorous formulations, categorize corresponding methodologies, and discuss benchmarks for each perspective. Moreover, we "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2209.08025","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2209.08025/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2209.08025","created_at":"2026-07-05T04:58:12.808324+00:00"},{"alias_kind":"arxiv_version","alias_value":"2209.08025v1","created_at":"2026-07-05T04:58:12.808324+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2209.08025","created_at":"2026-07-05T04:58:12.808324+00:00"},{"alias_kind":"pith_short_12","alias_value":"WLNNFXWWCEJE","created_at":"2026-07-05T04:58:12.808324+00:00"},{"alias_kind":"pith_short_16","alias_value":"WLNNFXWWCEJENOUT","created_at":"2026-07-05T04:58:12.808324+00:00"},{"alias_kind":"pith_short_8","alias_value":"WLNNFXWW","created_at":"2026-07-05T04:58:12.808324+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.30829","citing_title":"Joint Chance Constrained Safe-Optimal Control","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06992","citing_title":"Why Does Agentic Safety Fail to Generalize Across Tasks?","ref_index":116,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/WLNNFXWWCEJENOUTOYJJBITJP6","json":"https://pith.science/pith/WLNNFXWWCEJENOUTOYJJBITJP6.json","graph_json":"https://pith.science/api/pith-number/WLNNFXWWCEJENOUTOYJJBITJP6/graph.json","events_json":"https://pith.science/api/pith-number/WLNNFXWWCEJENOUTOYJJBITJP6/events.json","paper":"https://pith.science/paper/WLNNFXWW"},"agent_actions":{"view_html":"https://pith.science/pith/WLNNFXWWCEJENOUTOYJJBITJP6","download_json":"https://pith.science/pith/WLNNFXWWCEJENOUTOYJJBITJP6.json","view_paper":"https://pith.science/paper/WLNNFXWW","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2209.08025&json=true","fetch_graph":"https://pith.science/api/pith-number/WLNNFXWWCEJENOUTOYJJBITJP6/graph.json","fetch_events":"https://pith.science/api/pith-number/WLNNFXWWCEJENOUTOYJJBITJP6/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/WLNNFXWWCEJENOUTOYJJBITJP6/action/timestamp_anchor","attest_storage":"https://pith.science/pith/WLNNFXWWCEJENOUTOYJJBITJP6/action/storage_attestation","attest_author":"https://pith.science/pith/WLNNFXWWCEJENOUTOYJJBITJP6/action/author_attestation","sign_citation":"https://pith.science/pith/WLNNFXWWCEJENOUTOYJJBITJP6/action/citation_signature","submit_replication":"https://pith.science/pith/WLNNFXWWCEJENOUTOYJJBITJP6/action/replication_record"}},"created_at":"2026-07-05T04:58:12.808324+00:00","updated_at":"2026-07-05T04:58:12.808324+00:00"}