{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2026:2PLDSIEVBTE52HRZ5EXXXURZ5C","short_pith_number":"pith:2PLDSIEV","schema_version":"1.0","canonical_sha256":"d3d63920950cc9dd1e39e92f7bd239e8a4e22fad4d03f5ad1845e2c14c3360d9","source":{"kind":"arxiv","id":"2602.08690","version":2},"attestation_state":"computed","paper":{"title":"SoK: The Pitfalls of Deep Reinforcement Learning for Cybersecurity","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CR"],"primary_cat":"cs.LG","authors_text":"Chris Hicks, Elizabeth Bates, Fabio Pierazzi, Ilias Tsingenopoulos, Myles Foley, Sanyam Vyas, Shae McFadden, Vasilios Mavroudis","submitted_at":"2026-02-09T14:12:41Z","abstract_excerpt":"Deep Reinforcement Learning (DRL) has achieved remarkable success in domains requiring sequential decision-making, motivating its application to cybersecurity problems. However, transitioning DRL from laboratory simulations to bespoke cyber environments can introduce numerous issues. This is further exacerbated by the often adversarial, non-stationary, and partially-observable nature of most cybersecurity tasks. In this paper, we identify and systematize 11 methodological pitfalls that frequently occur in DRL for cybersecurity (DRL4Sec) literature across the stages of environment modeling, age"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2602.08690","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2026-02-09T14:12:41Z","cross_cats_sorted":["cs.CR"],"title_canon_sha256":"3346f31aed5202ffd94b5e6cad64b0f467abb0cfae3ff8f183c533919ed48769","abstract_canon_sha256":"39b44e959da28045a37652df1da90a340fa601dd70fba42165d4fa92e32936a8"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-06-23T03:13:54.149104Z","signature_b64":"WWEC+YZn257woe1hEGN9S2ihRd89ISIkmqej0tYFVcdhtH96TcOQkV6xlmoeyaDj3dAOrBHP+5bB4rWMCaTQAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"d3d63920950cc9dd1e39e92f7bd239e8a4e22fad4d03f5ad1845e2c14c3360d9","last_reissued_at":"2026-06-23T03:13:54.148607Z","signature_status":"signed_v1","first_computed_at":"2026-06-23T03:13:54.148607Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"SoK: The Pitfalls of Deep Reinforcement Learning for Cybersecurity","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CR"],"primary_cat":"cs.LG","authors_text":"Chris Hicks, Elizabeth Bates, Fabio Pierazzi, Ilias Tsingenopoulos, Myles Foley, Sanyam Vyas, Shae McFadden, Vasilios Mavroudis","submitted_at":"2026-02-09T14:12:41Z","abstract_excerpt":"Deep Reinforcement Learning (DRL) has achieved remarkable success in domains requiring sequential decision-making, motivating its application to cybersecurity problems. However, transitioning DRL from laboratory simulations to bespoke cyber environments can introduce numerous issues. This is further exacerbated by the often adversarial, non-stationary, and partially-observable nature of most cybersecurity tasks. In this paper, we identify and systematize 11 methodological pitfalls that frequently occur in DRL for cybersecurity (DRL4Sec) literature across the stages of environment modeling, age"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2602.08690","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2602.08690/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2602.08690","created_at":"2026-06-23T03:13:54.148678+00:00"},{"alias_kind":"arxiv_version","alias_value":"2602.08690v2","created_at":"2026-06-23T03:13:54.148678+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2602.08690","created_at":"2026-06-23T03:13:54.148678+00:00"},{"alias_kind":"pith_short_12","alias_value":"2PLDSIEVBTE5","created_at":"2026-06-23T03:13:54.148678+00:00"},{"alias_kind":"pith_short_16","alias_value":"2PLDSIEVBTE52HRZ","created_at":"2026-06-23T03:13:54.148678+00:00"},{"alias_kind":"pith_short_8","alias_value":"2PLDSIEV","created_at":"2026-06-23T03:13:54.148678+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2604.08805","citing_title":"Building Better Environments for Autonomous Cyber Defence","ref_index":48,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/2PLDSIEVBTE52HRZ5EXXXURZ5C","json":"https://pith.science/pith/2PLDSIEVBTE52HRZ5EXXXURZ5C.json","graph_json":"https://pith.science/api/pith-number/2PLDSIEVBTE52HRZ5EXXXURZ5C/graph.json","events_json":"https://pith.science/api/pith-number/2PLDSIEVBTE52HRZ5EXXXURZ5C/events.json","paper":"https://pith.science/paper/2PLDSIEV"},"agent_actions":{"view_html":"https://pith.science/pith/2PLDSIEVBTE52HRZ5EXXXURZ5C","download_json":"https://pith.science/pith/2PLDSIEVBTE52HRZ5EXXXURZ5C.json","view_paper":"https://pith.science/paper/2PLDSIEV","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2602.08690&json=true","fetch_graph":"https://pith.science/api/pith-number/2PLDSIEVBTE52HRZ5EXXXURZ5C/graph.json","fetch_events":"https://pith.science/api/pith-number/2PLDSIEVBTE52HRZ5EXXXURZ5C/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/2PLDSIEVBTE52HRZ5EXXXURZ5C/action/timestamp_anchor","attest_storage":"https://pith.science/pith/2PLDSIEVBTE52HRZ5EXXXURZ5C/action/storage_attestation","attest_author":"https://pith.science/pith/2PLDSIEVBTE52HRZ5EXXXURZ5C/action/author_attestation","sign_citation":"https://pith.science/pith/2PLDSIEVBTE52HRZ5EXXXURZ5C/action/citation_signature","submit_replication":"https://pith.science/pith/2PLDSIEVBTE52HRZ5EXXXURZ5C/action/replication_record"}},"created_at":"2026-06-23T03:13:54.148678+00:00","updated_at":"2026-06-23T03:13:54.148678+00:00"}