{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:2IP65T5DV3W7R2QOP5HJXHT4G5","short_pith_number":"pith:2IP65T5D","schema_version":"1.0","canonical_sha256":"d21feecfa3aeedf8ea0e7f4e9b9e7c376f69d7f150d005330d265755d68cf65a","source":{"kind":"arxiv","id":"2503.04262","version":1},"attestation_state":"computed","paper":{"title":"Guidelines for Applying RL and MARL in Cybersecurity Applications","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Adam Price, Alberto Caron, David Foster, Gregory Palmer, Ian Miles, Kez Smithson Whitehead, Sara Farmer, Stephen Pasteris, Vasilios Mavroudis","submitted_at":"2025-03-06T09:46:16Z","abstract_excerpt":"Reinforcement Learning (RL) and Multi-Agent Reinforcement Learning (MARL) have emerged as promising methodologies for addressing challenges in automated cyber defence (ACD). These techniques offer adaptive decision-making capabilities in high-dimensional, adversarial environments. This report provides a structured set of guidelines for cybersecurity professionals and researchers to assess the suitability of RL and MARL for specific use cases, considering factors such as explainability, exploration needs, and the complexity of multi-agent coordination. It also discusses key algorithmic approach"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2503.04262","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2025-03-06T09:46:16Z","cross_cats_sorted":[],"title_canon_sha256":"a617cdd22115c55fd772ea28bfe04805e625312b745381fb7daaa5fffadcf9e4","abstract_canon_sha256":"9f2ca01fc9d3ba199cce8e3aae2684c04c4aef5dd77c75f46219853323f8178a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:25:33.984358Z","signature_b64":"3z9aEgSGQvI9ogerdyjiX8Ot881B7sZOG7m938qwQTGHiumXqu+XBljyC3FcnzWhlcw+Kr9Iq8w/GeVDU+tXAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"d21feecfa3aeedf8ea0e7f4e9b9e7c376f69d7f150d005330d265755d68cf65a","last_reissued_at":"2026-07-05T10:25:33.983377Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:25:33.983377Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Guidelines for Applying RL and MARL in Cybersecurity Applications","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.AI","authors_text":"Adam Price, Alberto Caron, David Foster, Gregory Palmer, Ian Miles, Kez Smithson Whitehead, Sara Farmer, Stephen Pasteris, Vasilios Mavroudis","submitted_at":"2025-03-06T09:46:16Z","abstract_excerpt":"Reinforcement Learning (RL) and Multi-Agent Reinforcement Learning (MARL) have emerged as promising methodologies for addressing challenges in automated cyber defence (ACD). These techniques offer adaptive decision-making capabilities in high-dimensional, adversarial environments. This report provides a structured set of guidelines for cybersecurity professionals and researchers to assess the suitability of RL and MARL for specific use cases, considering factors such as explainability, exploration needs, and the complexity of multi-agent coordination. It also discusses key algorithmic approach"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2503.04262","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2503.04262/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2503.04262","created_at":"2026-07-05T10:25:33.983504+00:00"},{"alias_kind":"arxiv_version","alias_value":"2503.04262v1","created_at":"2026-07-05T10:25:33.983504+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2503.04262","created_at":"2026-07-05T10:25:33.983504+00:00"},{"alias_kind":"pith_short_12","alias_value":"2IP65T5DV3W7","created_at":"2026-07-05T10:25:33.983504+00:00"},{"alias_kind":"pith_short_16","alias_value":"2IP65T5DV3W7R2QO","created_at":"2026-07-05T10:25:33.983504+00:00"},{"alias_kind":"pith_short_8","alias_value":"2IP65T5D","created_at":"2026-07-05T10:25:33.983504+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.00741","citing_title":"Self-Adaptive Multi-Agent LLM-Based Security Pattern Selection for IoT Systems","ref_index":34,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/2IP65T5DV3W7R2QOP5HJXHT4G5","json":"https://pith.science/pith/2IP65T5DV3W7R2QOP5HJXHT4G5.json","graph_json":"https://pith.science/api/pith-number/2IP65T5DV3W7R2QOP5HJXHT4G5/graph.json","events_json":"https://pith.science/api/pith-number/2IP65T5DV3W7R2QOP5HJXHT4G5/events.json","paper":"https://pith.science/paper/2IP65T5D"},"agent_actions":{"view_html":"https://pith.science/pith/2IP65T5DV3W7R2QOP5HJXHT4G5","download_json":"https://pith.science/pith/2IP65T5DV3W7R2QOP5HJXHT4G5.json","view_paper":"https://pith.science/paper/2IP65T5D","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2503.04262&json=true","fetch_graph":"https://pith.science/api/pith-number/2IP65T5DV3W7R2QOP5HJXHT4G5/graph.json","fetch_events":"https://pith.science/api/pith-number/2IP65T5DV3W7R2QOP5HJXHT4G5/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/2IP65T5DV3W7R2QOP5HJXHT4G5/action/timestamp_anchor","attest_storage":"https://pith.science/pith/2IP65T5DV3W7R2QOP5HJXHT4G5/action/storage_attestation","attest_author":"https://pith.science/pith/2IP65T5DV3W7R2QOP5HJXHT4G5/action/author_attestation","sign_citation":"https://pith.science/pith/2IP65T5DV3W7R2QOP5HJXHT4G5/action/citation_signature","submit_replication":"https://pith.science/pith/2IP65T5DV3W7R2QOP5HJXHT4G5/action/replication_record"}},"created_at":"2026-07-05T10:25:33.983504+00:00","updated_at":"2026-07-05T10:25:33.983504+00:00"}