{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:6QMW5RN7SHVCFSOVZG36Z3NLUC","short_pith_number":"pith:6QMW5RN7","schema_version":"1.0","canonical_sha256":"f4196ec5bf91ea22c9d5c9b7ecedaba082bda489c63153b8778b1fe2e14b9c4c","source":{"kind":"arxiv","id":"2101.08452","version":1},"attestation_state":"computed","paper":{"title":"Robust Reinforcement Learning on State Observations with Learned Optimal Adversary","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","stat.ML"],"primary_cat":"cs.LG","authors_text":"Cho-Jui Hsieh, Duane Boning, Hongge Chen, Huan Zhang","submitted_at":"2021-01-21T05:38:52Z","abstract_excerpt":"We study the robustness of reinforcement learning (RL) with adversarially perturbed state observations, which aligns with the setting of many adversarial attacks to deep reinforcement learning (DRL) and is also important for rolling out real-world RL agent under unpredictable sensing noise. With a fixed agent policy, we demonstrate that an optimal adversary to perturb state observations can be found, which is guaranteed to obtain the worst case agent reward. For DRL settings, this leads to a novel empirical adversarial attack to RL agents via a learned adversary that is much stronger than prev"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2101.08452","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2021-01-21T05:38:52Z","cross_cats_sorted":["cs.AI","stat.ML"],"title_canon_sha256":"4a7b8d95b814104d62bf221357700ff50c18d91b81e4814c18fa19fc6d629bff","abstract_canon_sha256":"da2f18d06930f2e117bd3e0147f0e8d04d9b04c795851c22ad461408baa066c8"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T02:08:35.428150Z","signature_b64":"F6WLKjE5IncpIELwKjEcaBsX1z+QBK4ROzs9GeTdcThng0/+EDtxM4ej9TkNbo2RpcPFvYJVWpE+C2/S4fu5Cg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f4196ec5bf91ea22c9d5c9b7ecedaba082bda489c63153b8778b1fe2e14b9c4c","last_reissued_at":"2026-07-05T02:08:35.427716Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T02:08:35.427716Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Robust Reinforcement Learning on State Observations with Learned Optimal Adversary","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","stat.ML"],"primary_cat":"cs.LG","authors_text":"Cho-Jui Hsieh, Duane Boning, Hongge Chen, Huan Zhang","submitted_at":"2021-01-21T05:38:52Z","abstract_excerpt":"We study the robustness of reinforcement learning (RL) with adversarially perturbed state observations, which aligns with the setting of many adversarial attacks to deep reinforcement learning (DRL) and is also important for rolling out real-world RL agent under unpredictable sensing noise. With a fixed agent policy, we demonstrate that an optimal adversary to perturb state observations can be found, which is guaranteed to obtain the worst case agent reward. For DRL settings, this leads to a novel empirical adversarial attack to RL agents via a learned adversary that is much stronger than prev"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2101.08452","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2101.08452/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2101.08452","created_at":"2026-07-05T02:08:35.427774+00:00"},{"alias_kind":"arxiv_version","alias_value":"2101.08452v1","created_at":"2026-07-05T02:08:35.427774+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2101.08452","created_at":"2026-07-05T02:08:35.427774+00:00"},{"alias_kind":"pith_short_12","alias_value":"6QMW5RN7SHVC","created_at":"2026-07-05T02:08:35.427774+00:00"},{"alias_kind":"pith_short_16","alias_value":"6QMW5RN7SHVCFSOV","created_at":"2026-07-05T02:08:35.427774+00:00"},{"alias_kind":"pith_short_8","alias_value":"6QMW5RN7","created_at":"2026-07-05T02:08:35.427774+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.18024","citing_title":"Interaction-Breaking Adversarial Learning Framework for Robust Multi-Agent Reinforcement Learning","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2606.29012","citing_title":"The Game Changer Problem: Controlling Equilibria with Discrete Rewards","ref_index":141,"is_internal_anchor":false},{"citing_arxiv_id":"2502.02844","citing_title":"Wolfpack Adversarial Attack for Robust Multi-Agent Reinforcement Learning","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18024","citing_title":"Interaction-Breaking Adversarial Learning Framework for Robust Multi-Agent Reinforcement Learning","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2604.10974","citing_title":"Robust Adversarial Policy Optimization Under Dynamics Uncertainty","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2605.03125","citing_title":"Taming the Curses of Multiagency in Robust Markov Games with Large State Space through Linear Function Approximation","ref_index":25,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/6QMW5RN7SHVCFSOVZG36Z3NLUC","json":"https://pith.science/pith/6QMW5RN7SHVCFSOVZG36Z3NLUC.json","graph_json":"https://pith.science/api/pith-number/6QMW5RN7SHVCFSOVZG36Z3NLUC/graph.json","events_json":"https://pith.science/api/pith-number/6QMW5RN7SHVCFSOVZG36Z3NLUC/events.json","paper":"https://pith.science/paper/6QMW5RN7"},"agent_actions":{"view_html":"https://pith.science/pith/6QMW5RN7SHVCFSOVZG36Z3NLUC","download_json":"https://pith.science/pith/6QMW5RN7SHVCFSOVZG36Z3NLUC.json","view_paper":"https://pith.science/paper/6QMW5RN7","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2101.08452&json=true","fetch_graph":"https://pith.science/api/pith-number/6QMW5RN7SHVCFSOVZG36Z3NLUC/graph.json","fetch_events":"https://pith.science/api/pith-number/6QMW5RN7SHVCFSOVZG36Z3NLUC/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/6QMW5RN7SHVCFSOVZG36Z3NLUC/action/timestamp_anchor","attest_storage":"https://pith.science/pith/6QMW5RN7SHVCFSOVZG36Z3NLUC/action/storage_attestation","attest_author":"https://pith.science/pith/6QMW5RN7SHVCFSOVZG36Z3NLUC/action/author_attestation","sign_citation":"https://pith.science/pith/6QMW5RN7SHVCFSOVZG36Z3NLUC/action/citation_signature","submit_replication":"https://pith.science/pith/6QMW5RN7SHVCFSOVZG36Z3NLUC/action/replication_record"}},"created_at":"2026-07-05T02:08:35.427774+00:00","updated_at":"2026-07-05T02:08:35.427774+00:00"}