{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2022:SKQBG74GXGNKMBCL22CMAP235O","short_pith_number":"pith:SKQBG74G","canonical_record":{"source":{"id":"2202.09514","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-sa/4.0/","primary_cat":"cs.LG","submitted_at":"2022-02-19T03:44:05Z","cross_cats_sorted":["cs.AI","cs.GT"],"title_canon_sha256":"c3370ec39faa0ed1b63ffe973344df22917ff4251ce4f1a20d934458bea2a1ed","abstract_canon_sha256":"89394486162ba0e4a831a0aab5e2b0bf45c583608eb0ddccfe1e25af2c22a483"},"schema_version":"1.0"},"canonical_sha256":"92a0137f86b99aa6044bd684c03f5beb9c02273c6a6f7c9682d0b4b170701230","source":{"kind":"arxiv","id":"2202.09514","version":2},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2202.09514","created_at":"2026-07-05T05:00:40Z"},{"alias_kind":"arxiv_version","alias_value":"2202.09514v2","created_at":"2026-07-05T05:00:40Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2202.09514","created_at":"2026-07-05T05:00:40Z"},{"alias_kind":"pith_short_12","alias_value":"SKQBG74GXGNK","created_at":"2026-07-05T05:00:40Z"},{"alias_kind":"pith_short_16","alias_value":"SKQBG74GXGNKMBCL","created_at":"2026-07-05T05:00:40Z"},{"alias_kind":"pith_short_8","alias_value":"SKQBG74G","created_at":"2026-07-05T05:00:40Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2022:SKQBG74GXGNKMBCL22CMAP235O","target":"record","payload":{"canonical_record":{"source":{"id":"2202.09514","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-sa/4.0/","primary_cat":"cs.LG","submitted_at":"2022-02-19T03:44:05Z","cross_cats_sorted":["cs.AI","cs.GT"],"title_canon_sha256":"c3370ec39faa0ed1b63ffe973344df22917ff4251ce4f1a20d934458bea2a1ed","abstract_canon_sha256":"89394486162ba0e4a831a0aab5e2b0bf45c583608eb0ddccfe1e25af2c22a483"},"schema_version":"1.0"},"canonical_sha256":"92a0137f86b99aa6044bd684c03f5beb9c02273c6a6f7c9682d0b4b170701230","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:00:40.428273Z","signature_b64":"UOYUxIBudVVOIWOGmgxq1m6SX5zc0io8tbvIvWDjL4wlsM5CnZ/oR7YP4f+qnXo+hOOSBAw7qp8Nlz0Y5tMEBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"92a0137f86b99aa6044bd684c03f5beb9c02273c6a6f7c9682d0b4b170701230","last_reissued_at":"2026-07-05T05:00:40.427801Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:00:40.427801Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2202.09514","source_version":2,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T05:00:40Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"JFAsmyYq6cBX9PM89/jZRrjEfuf7UhJLqNaBle9Vao7GFbwOs2okN9LE84oYp6ADMEv3vG2nYhOLxrPF5h1uDA==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-05T23:12:26.726945Z"},"content_sha256":"e74f621b59a0650279ae3bfc38fdb76e4e77380d0c94a4a68b8b4c05718eee9d","schema_version":"1.0","event_id":"sha256:e74f621b59a0650279ae3bfc38fdb76e4e77380d0c94a4a68b8b4c05718eee9d"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2022:SKQBG74GXGNKMBCL22CMAP235O","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Robust Reinforcement Learning as a Stackelberg Game via Adaptively-Regularized Adversarial Training","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":["cs.AI","cs.GT"],"primary_cat":"cs.LG","authors_text":"Ding Zhao, Fei Fang, Mengdi Xu, Peide Huang","submitted_at":"2022-02-19T03:44:05Z","abstract_excerpt":"Robust Reinforcement Learning (RL) focuses on improving performances under model errors or adversarial attacks, which facilitates the real-life deployment of RL agents. Robust Adversarial Reinforcement Learning (RARL) is one of the most popular frameworks for robust RL. However, most of the existing literature models RARL as a zero-sum simultaneous game with Nash equilibrium as the solution concept, which could overlook the sequential nature of RL deployments, produce overly conservative agents, and induce training instability. In this paper, we introduce a novel hierarchical formulation of ro"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2202.09514","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2202.09514/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T05:00:40Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"REtTVO0cYvImPzse9f/0Y0pPMgS+EQKDIvTsDu8mPnPtU2jK0yXRcBs/T933h/xaeUzWYM1xhXym2er0kMXQBw==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-05T23:12:26.727457Z"},"content_sha256":"7a528563a82945a6f160dcd0283daa7ee67b64ae7d6dd96d44f6e9601b18be57","schema_version":"1.0","event_id":"sha256:7a528563a82945a6f160dcd0283daa7ee67b64ae7d6dd96d44f6e9601b18be57"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/SKQBG74GXGNKMBCL22CMAP235O/bundle.json","state_url":"https://pith.science/pith/SKQBG74GXGNKMBCL22CMAP235O/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/SKQBG74GXGNKMBCL22CMAP235O/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-05T23:12:26Z","links":{"resolver":"https://pith.science/pith/SKQBG74GXGNKMBCL22CMAP235O","bundle":"https://pith.science/pith/SKQBG74GXGNKMBCL22CMAP235O/bundle.json","state":"https://pith.science/pith/SKQBG74GXGNKMBCL22CMAP235O/state.json","well_known_bundle":"https://pith.science/.well-known/pith/SKQBG74GXGNKMBCL22CMAP235O/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2022:SKQBG74GXGNKMBCL22CMAP235O","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"89394486162ba0e4a831a0aab5e2b0bf45c583608eb0ddccfe1e25af2c22a483","cross_cats_sorted":["cs.AI","cs.GT"],"license":"http://creativecommons.org/licenses/by-sa/4.0/","primary_cat":"cs.LG","submitted_at":"2022-02-19T03:44:05Z","title_canon_sha256":"c3370ec39faa0ed1b63ffe973344df22917ff4251ce4f1a20d934458bea2a1ed"},"schema_version":"1.0","source":{"id":"2202.09514","kind":"arxiv","version":2}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2202.09514","created_at":"2026-07-05T05:00:40Z"},{"alias_kind":"arxiv_version","alias_value":"2202.09514v2","created_at":"2026-07-05T05:00:40Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2202.09514","created_at":"2026-07-05T05:00:40Z"},{"alias_kind":"pith_short_12","alias_value":"SKQBG74GXGNK","created_at":"2026-07-05T05:00:40Z"},{"alias_kind":"pith_short_16","alias_value":"SKQBG74GXGNKMBCL","created_at":"2026-07-05T05:00:40Z"},{"alias_kind":"pith_short_8","alias_value":"SKQBG74G","created_at":"2026-07-05T05:00:40Z"}],"graph_snapshots":[{"event_id":"sha256:7a528563a82945a6f160dcd0283daa7ee67b64ae7d6dd96d44f6e9601b18be57","target":"graph","created_at":"2026-07-05T05:00:40Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2202.09514/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Robust Reinforcement Learning (RL) focuses on improving performances under model errors or adversarial attacks, which facilitates the real-life deployment of RL agents. Robust Adversarial Reinforcement Learning (RARL) is one of the most popular frameworks for robust RL. However, most of the existing literature models RARL as a zero-sum simultaneous game with Nash equilibrium as the solution concept, which could overlook the sequential nature of RL deployments, produce overly conservative agents, and induce training instability. In this paper, we introduce a novel hierarchical formulation of ro","authors_text":"Ding Zhao, Fei Fang, Mengdi Xu, Peide Huang","cross_cats":["cs.AI","cs.GT"],"headline":"","license":"http://creativecommons.org/licenses/by-sa/4.0/","primary_cat":"cs.LG","submitted_at":"2022-02-19T03:44:05Z","title":"Robust Reinforcement Learning as a Stackelberg Game via Adaptively-Regularized Adversarial Training"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2202.09514","kind":"arxiv","version":2},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:e74f621b59a0650279ae3bfc38fdb76e4e77380d0c94a4a68b8b4c05718eee9d","target":"record","created_at":"2026-07-05T05:00:40Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"89394486162ba0e4a831a0aab5e2b0bf45c583608eb0ddccfe1e25af2c22a483","cross_cats_sorted":["cs.AI","cs.GT"],"license":"http://creativecommons.org/licenses/by-sa/4.0/","primary_cat":"cs.LG","submitted_at":"2022-02-19T03:44:05Z","title_canon_sha256":"c3370ec39faa0ed1b63ffe973344df22917ff4251ce4f1a20d934458bea2a1ed"},"schema_version":"1.0","source":{"id":"2202.09514","kind":"arxiv","version":2}},"canonical_sha256":"92a0137f86b99aa6044bd684c03f5beb9c02273c6a6f7c9682d0b4b170701230","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"92a0137f86b99aa6044bd684c03f5beb9c02273c6a6f7c9682d0b4b170701230","first_computed_at":"2026-07-05T05:00:40.427801Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T05:00:40.427801Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"UOYUxIBudVVOIWOGmgxq1m6SX5zc0io8tbvIvWDjL4wlsM5CnZ/oR7YP4f+qnXo+hOOSBAw7qp8Nlz0Y5tMEBQ==","signature_status":"signed_v1","signed_at":"2026-07-05T05:00:40.428273Z","signed_message":"canonical_sha256_bytes"},"source_id":"2202.09514","source_kind":"arxiv","source_version":2}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:e74f621b59a0650279ae3bfc38fdb76e4e77380d0c94a4a68b8b4c05718eee9d","sha256:7a528563a82945a6f160dcd0283daa7ee67b64ae7d6dd96d44f6e9601b18be57"],"state_sha256":"7ba2983a76ff6a098ff22580dc141e223d0d79cf50789fda1b5e2a8a54688adc"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"UxJlhpnchtzt0ePMYRK+tyESx6D5vONN3B9mrnhIAkoUO5hXakARLIwjSuNF4fk3eKvKXQbsmQkD6jaIKGDBAQ==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-05T23:12:26.731304Z","bundle_sha256":"5a4ac69139abe469a46ac01fbfc3b10c8802ac06fc89b690ec2a72191fa8dd0a"}}