{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2022:TA5EOQM2F32MU4GCCGSRAERXYP","short_pith_number":"pith:TA5EOQM2","canonical_record":{"source":{"id":"2205.07344","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2022-05-15T17:35:17Z","cross_cats_sorted":[],"title_canon_sha256":"b619f04a03fa7639a8abd36b1fc23ff9ee9364a7450c9b52fc8a001cc5c6dced","abstract_canon_sha256":"5d1d3bb420a3aaa1016ca37e22562fc70dc413bc8c75b2ab3de760c786632c7d"},"schema_version":"1.0"},"canonical_sha256":"983a47419a2ef4ca70c211a5101237c3c70bcebd7c78fdeba59d98e9d78484e7","source":{"kind":"arxiv","id":"2205.07344","version":1},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2205.07344","created_at":"2026-07-05T04:23:28Z"},{"alias_kind":"arxiv_version","alias_value":"2205.07344v1","created_at":"2026-07-05T04:23:28Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2205.07344","created_at":"2026-07-05T04:23:28Z"},{"alias_kind":"pith_short_12","alias_value":"TA5EOQM2F32M","created_at":"2026-07-05T04:23:28Z"},{"alias_kind":"pith_short_16","alias_value":"TA5EOQM2F32MU4GC","created_at":"2026-07-05T04:23:28Z"},{"alias_kind":"pith_short_8","alias_value":"TA5EOQM2","created_at":"2026-07-05T04:23:28Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2022:TA5EOQM2F32MU4GCCGSRAERXYP","target":"record","payload":{"canonical_record":{"source":{"id":"2205.07344","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2022-05-15T17:35:17Z","cross_cats_sorted":[],"title_canon_sha256":"b619f04a03fa7639a8abd36b1fc23ff9ee9364a7450c9b52fc8a001cc5c6dced","abstract_canon_sha256":"5d1d3bb420a3aaa1016ca37e22562fc70dc413bc8c75b2ab3de760c786632c7d"},"schema_version":"1.0"},"canonical_sha256":"983a47419a2ef4ca70c211a5101237c3c70bcebd7c78fdeba59d98e9d78484e7","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T04:23:28.011133Z","signature_b64":"+H6ys07DLtohUfX/NLHdpMcMC5gsuLUum9EtZUI362YDV+Zu5OZXyEqlb8kyLDK/8GdETw72Y/oE0axf9RXOAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"983a47419a2ef4ca70c211a5101237c3c70bcebd7c78fdeba59d98e9d78484e7","last_reissued_at":"2026-07-05T04:23:28.010662Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T04:23:28.010662Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2205.07344","source_version":1,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T04:23:28Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"psvGnLSuneBo9W1OQV0UvKDYesoHMwkve07s+hERF+53gwRX6Q2Bj1kS3DVDew8A251HfRwEuYncNE24tU0jBQ==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-12T21:25:42.718492Z"},"content_sha256":"330b91f1fb31bf9b0a6623d394dab72f32c1b278d32edcdff48d89fde6391aad","schema_version":"1.0","event_id":"sha256:330b91f1fb31bf9b0a6623d394dab72f32c1b278d32edcdff48d89fde6391aad"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2022:TA5EOQM2F32MU4GCCGSRAERXYP","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Policy Gradient Method For Robust Reinforcement Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Shaofeng Zou, Yue Wang","submitted_at":"2022-05-15T17:35:17Z","abstract_excerpt":"This paper develops the first policy gradient method with global optimality guarantee and complexity analysis for robust reinforcement learning under model mismatch. Robust reinforcement learning is to learn a policy robust to model mismatch between simulator and real environment. We first develop the robust policy (sub-)gradient, which is applicable for any differentiable parametric policy class. We show that the proposed robust policy gradient method converges to the global optimum asymptotically under direct policy parameterization. We further develop a smoothed robust policy gradient metho"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2205.07344","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2205.07344/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T04:23:28Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"y/XGsuxvvNgkJeu6Anop+lWuA1jA4OsfJBFdbqaEvxjIdlYtCY74KknDNHNe4mbMmoW3kdsitzeBZyYcZwHvDg==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-12T21:25:42.718999Z"},"content_sha256":"022b9d3e24ec7da6528dd08244d69ce1e04fa175e23e39b895ac3aa128ff1877","schema_version":"1.0","event_id":"sha256:022b9d3e24ec7da6528dd08244d69ce1e04fa175e23e39b895ac3aa128ff1877"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/TA5EOQM2F32MU4GCCGSRAERXYP/bundle.json","state_url":"https://pith.science/pith/TA5EOQM2F32MU4GCCGSRAERXYP/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/TA5EOQM2F32MU4GCCGSRAERXYP/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-12T21:25:42Z","links":{"resolver":"https://pith.science/pith/TA5EOQM2F32MU4GCCGSRAERXYP","bundle":"https://pith.science/pith/TA5EOQM2F32MU4GCCGSRAERXYP/bundle.json","state":"https://pith.science/pith/TA5EOQM2F32MU4GCCGSRAERXYP/state.json","well_known_bundle":"https://pith.science/.well-known/pith/TA5EOQM2F32MU4GCCGSRAERXYP/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2022:TA5EOQM2F32MU4GCCGSRAERXYP","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"5d1d3bb420a3aaa1016ca37e22562fc70dc413bc8c75b2ab3de760c786632c7d","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2022-05-15T17:35:17Z","title_canon_sha256":"b619f04a03fa7639a8abd36b1fc23ff9ee9364a7450c9b52fc8a001cc5c6dced"},"schema_version":"1.0","source":{"id":"2205.07344","kind":"arxiv","version":1}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2205.07344","created_at":"2026-07-05T04:23:28Z"},{"alias_kind":"arxiv_version","alias_value":"2205.07344v1","created_at":"2026-07-05T04:23:28Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2205.07344","created_at":"2026-07-05T04:23:28Z"},{"alias_kind":"pith_short_12","alias_value":"TA5EOQM2F32M","created_at":"2026-07-05T04:23:28Z"},{"alias_kind":"pith_short_16","alias_value":"TA5EOQM2F32MU4GC","created_at":"2026-07-05T04:23:28Z"},{"alias_kind":"pith_short_8","alias_value":"TA5EOQM2","created_at":"2026-07-05T04:23:28Z"}],"graph_snapshots":[{"event_id":"sha256:022b9d3e24ec7da6528dd08244d69ce1e04fa175e23e39b895ac3aa128ff1877","target":"graph","created_at":"2026-07-05T04:23:28Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2205.07344/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"This paper develops the first policy gradient method with global optimality guarantee and complexity analysis for robust reinforcement learning under model mismatch. Robust reinforcement learning is to learn a policy robust to model mismatch between simulator and real environment. We first develop the robust policy (sub-)gradient, which is applicable for any differentiable parametric policy class. We show that the proposed robust policy gradient method converges to the global optimum asymptotically under direct policy parameterization. We further develop a smoothed robust policy gradient metho","authors_text":"Shaofeng Zou, Yue Wang","cross_cats":[],"headline":"","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2022-05-15T17:35:17Z","title":"Policy Gradient Method For Robust Reinforcement Learning"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2205.07344","kind":"arxiv","version":1},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:330b91f1fb31bf9b0a6623d394dab72f32c1b278d32edcdff48d89fde6391aad","target":"record","created_at":"2026-07-05T04:23:28Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"5d1d3bb420a3aaa1016ca37e22562fc70dc413bc8c75b2ab3de760c786632c7d","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2022-05-15T17:35:17Z","title_canon_sha256":"b619f04a03fa7639a8abd36b1fc23ff9ee9364a7450c9b52fc8a001cc5c6dced"},"schema_version":"1.0","source":{"id":"2205.07344","kind":"arxiv","version":1}},"canonical_sha256":"983a47419a2ef4ca70c211a5101237c3c70bcebd7c78fdeba59d98e9d78484e7","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"983a47419a2ef4ca70c211a5101237c3c70bcebd7c78fdeba59d98e9d78484e7","first_computed_at":"2026-07-05T04:23:28.010662Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T04:23:28.010662Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"+H6ys07DLtohUfX/NLHdpMcMC5gsuLUum9EtZUI362YDV+Zu5OZXyEqlb8kyLDK/8GdETw72Y/oE0axf9RXOAA==","signature_status":"signed_v1","signed_at":"2026-07-05T04:23:28.011133Z","signed_message":"canonical_sha256_bytes"},"source_id":"2205.07344","source_kind":"arxiv","source_version":1}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:330b91f1fb31bf9b0a6623d394dab72f32c1b278d32edcdff48d89fde6391aad","sha256:022b9d3e24ec7da6528dd08244d69ce1e04fa175e23e39b895ac3aa128ff1877"],"state_sha256":"53a37713ed37956de9c453d2eeb7bd974c970ba221912a40254267d916ef8294"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"v3A9U3pPRGelVOuk8DyRy6OzDrXRQITW5FMwDg/FKoFBynLvhdwRWdtSGhEcTnjVRrHGUCrI9hnq0uIffNKOBA==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-12T21:25:42.723972Z","bundle_sha256":"fe062b10843df553c0bcc6b1bb4ffc87fadf213a172f085430ff766fe812722c"}}