{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:4BTHXRMD5OXCQRF2VHLJBAFQ7Z","short_pith_number":"pith:4BTHXRMD","schema_version":"1.0","canonical_sha256":"e0667bc583ebae2844baa9d69080b0fe47c3b9fec4aec3e273611816a8cf0d70","source":{"kind":"arxiv","id":"2207.09597","version":2},"attestation_state":"computed","paper":{"title":"Feasible Adversarial Robust Reinforcement Learning for Underspecified Environments","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.GT"],"primary_cat":"cs.LG","authors_text":"JB Lanier, Pierre Baldi, Roy Fox, Stephen McAleer","submitted_at":"2022-07-19T23:57:51Z","abstract_excerpt":"Robust reinforcement learning (RL) considers the problem of learning policies that perform well in the worst case among a set of possible environment parameter values. In real-world environments, choosing the set of possible values for robust RL can be a difficult task. When that set is specified too narrowly, the agent will be left vulnerable to reasonable parameter values unaccounted for. When specified too broadly, the agent will be too cautious. In this paper, we propose Feasible Adversarial Robust RL (FARR), a novel problem formulation and objective for automatically determining the set o"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2207.09597","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2022-07-19T23:57:51Z","cross_cats_sorted":["cs.AI","cs.GT"],"title_canon_sha256":"09cfd96234b2d2c83bfff4e52659cca4b8d16caab1803771c8abea0638459289","abstract_canon_sha256":"cd4485cfe089291d95a3d2417ab35d55ec8f1f37854099f3d46966d915ca9946"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:03:04.170891Z","signature_b64":"baWMi5dymgQa/0HhzpVyZ+5pZqi+Gqd4y/kRnsyNv2jTHMhRXMRAiOgP6NdyDuXKotm2+i/LhR1DaSR4Kc1VAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"e0667bc583ebae2844baa9d69080b0fe47c3b9fec4aec3e273611816a8cf0d70","last_reissued_at":"2026-07-05T05:03:04.170369Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:03:04.170369Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Feasible Adversarial Robust Reinforcement Learning for Underspecified Environments","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.GT"],"primary_cat":"cs.LG","authors_text":"JB Lanier, Pierre Baldi, Roy Fox, Stephen McAleer","submitted_at":"2022-07-19T23:57:51Z","abstract_excerpt":"Robust reinforcement learning (RL) considers the problem of learning policies that perform well in the worst case among a set of possible environment parameter values. In real-world environments, choosing the set of possible values for robust RL can be a difficult task. When that set is specified too narrowly, the agent will be left vulnerable to reasonable parameter values unaccounted for. When specified too broadly, the agent will be too cautious. In this paper, we propose Feasible Adversarial Robust RL (FARR), a novel problem formulation and objective for automatically determining the set o"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2207.09597","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2207.09597/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2207.09597","created_at":"2026-07-05T05:03:04.170419+00:00"},{"alias_kind":"arxiv_version","alias_value":"2207.09597v2","created_at":"2026-07-05T05:03:04.170419+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2207.09597","created_at":"2026-07-05T05:03:04.170419+00:00"},{"alias_kind":"pith_short_12","alias_value":"4BTHXRMD5OXC","created_at":"2026-07-05T05:03:04.170419+00:00"},{"alias_kind":"pith_short_16","alias_value":"4BTHXRMD5OXCQRF2","created_at":"2026-07-05T05:03:04.170419+00:00"},{"alias_kind":"pith_short_8","alias_value":"4BTHXRMD","created_at":"2026-07-05T05:03:04.170419+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/4BTHXRMD5OXCQRF2VHLJBAFQ7Z","json":"https://pith.science/pith/4BTHXRMD5OXCQRF2VHLJBAFQ7Z.json","graph_json":"https://pith.science/api/pith-number/4BTHXRMD5OXCQRF2VHLJBAFQ7Z/graph.json","events_json":"https://pith.science/api/pith-number/4BTHXRMD5OXCQRF2VHLJBAFQ7Z/events.json","paper":"https://pith.science/paper/4BTHXRMD"},"agent_actions":{"view_html":"https://pith.science/pith/4BTHXRMD5OXCQRF2VHLJBAFQ7Z","download_json":"https://pith.science/pith/4BTHXRMD5OXCQRF2VHLJBAFQ7Z.json","view_paper":"https://pith.science/paper/4BTHXRMD","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2207.09597&json=true","fetch_graph":"https://pith.science/api/pith-number/4BTHXRMD5OXCQRF2VHLJBAFQ7Z/graph.json","fetch_events":"https://pith.science/api/pith-number/4BTHXRMD5OXCQRF2VHLJBAFQ7Z/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/4BTHXRMD5OXCQRF2VHLJBAFQ7Z/action/timestamp_anchor","attest_storage":"https://pith.science/pith/4BTHXRMD5OXCQRF2VHLJBAFQ7Z/action/storage_attestation","attest_author":"https://pith.science/pith/4BTHXRMD5OXCQRF2VHLJBAFQ7Z/action/author_attestation","sign_citation":"https://pith.science/pith/4BTHXRMD5OXCQRF2VHLJBAFQ7Z/action/citation_signature","submit_replication":"https://pith.science/pith/4BTHXRMD5OXCQRF2VHLJBAFQ7Z/action/replication_record"}},"created_at":"2026-07-05T05:03:04.170419+00:00","updated_at":"2026-07-05T05:03:04.170419+00:00"}