{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:BENDYJ7OEFKW7JM5ZFJ5VHYU2V","short_pith_number":"pith:BENDYJ7O","schema_version":"1.0","canonical_sha256":"091a3c27ee21556fa59dc953da9f14d552bdca2358809db032722f9e19706b70","source":{"kind":"arxiv","id":"2310.05939","version":1},"attestation_state":"computed","paper":{"title":"Learning Cyber Defence Tactics from Scratch with Multi-Agent Reinforcement Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CR","authors_text":"Jacob Wiebe, Li Li, Ranwa Al Mallah","submitted_at":"2023-08-25T14:07:50Z","abstract_excerpt":"Recent advancements in deep learning techniques have opened new possibilities for designing solutions for autonomous cyber defence. Teams of intelligent agents in computer network defence roles may reveal promising avenues to safeguard cyber and kinetic assets. In a simulated game environment, agents are evaluated on their ability to jointly mitigate attacker activity in host-based defence scenarios. Defender systems are evaluated against heuristic attackers with the goals of compromising network confidentiality, integrity, and availability. Value-based Independent Learning and Centralized Tra"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2310.05939","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CR","submitted_at":"2023-08-25T14:07:50Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"30d955bcfc3560f2b832348a1350052558003db78e8c146ec0f638d2f6742a04","abstract_canon_sha256":"29f7e37b6a3aa9b323f179973d56d08f5895df979074b5a1570a636c4b186a1f"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:59:02.097235Z","signature_b64":"c+X06QlqYSiFS7HfR15vSomoGXLFLNs964IN2TjlxjtfRTihDo5cBV25oCwjq9qCWuC29W8pHZaOSo0ERCKFDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"091a3c27ee21556fa59dc953da9f14d552bdca2358809db032722f9e19706b70","last_reissued_at":"2026-07-05T06:59:02.096676Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:59:02.096676Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Learning Cyber Defence Tactics from Scratch with Multi-Agent Reinforcement Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CR","authors_text":"Jacob Wiebe, Li Li, Ranwa Al Mallah","submitted_at":"2023-08-25T14:07:50Z","abstract_excerpt":"Recent advancements in deep learning techniques have opened new possibilities for designing solutions for autonomous cyber defence. Teams of intelligent agents in computer network defence roles may reveal promising avenues to safeguard cyber and kinetic assets. In a simulated game environment, agents are evaluated on their ability to jointly mitigate attacker activity in host-based defence scenarios. Defender systems are evaluated against heuristic attackers with the goals of compromising network confidentiality, integrity, and availability. Value-based Independent Learning and Centralized Tra"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2310.05939","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2310.05939/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2310.05939","created_at":"2026-07-05T06:59:02.096733+00:00"},{"alias_kind":"arxiv_version","alias_value":"2310.05939v1","created_at":"2026-07-05T06:59:02.096733+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2310.05939","created_at":"2026-07-05T06:59:02.096733+00:00"},{"alias_kind":"pith_short_12","alias_value":"BENDYJ7OEFKW","created_at":"2026-07-05T06:59:02.096733+00:00"},{"alias_kind":"pith_short_16","alias_value":"BENDYJ7OEFKW7JM5","created_at":"2026-07-05T06:59:02.096733+00:00"},{"alias_kind":"pith_short_8","alias_value":"BENDYJ7O","created_at":"2026-07-05T06:59:02.096733+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2505.11708","citing_title":"Unveiling the Black Box: A Multi-Layer Framework for Explaining Reinforcement Learning-Based Cyber Agents","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2507.15163","citing_title":"Adaptive Network Security Policies via Belief Aggregation and Rollout","ref_index":39,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/BENDYJ7OEFKW7JM5ZFJ5VHYU2V","json":"https://pith.science/pith/BENDYJ7OEFKW7JM5ZFJ5VHYU2V.json","graph_json":"https://pith.science/api/pith-number/BENDYJ7OEFKW7JM5ZFJ5VHYU2V/graph.json","events_json":"https://pith.science/api/pith-number/BENDYJ7OEFKW7JM5ZFJ5VHYU2V/events.json","paper":"https://pith.science/paper/BENDYJ7O"},"agent_actions":{"view_html":"https://pith.science/pith/BENDYJ7OEFKW7JM5ZFJ5VHYU2V","download_json":"https://pith.science/pith/BENDYJ7OEFKW7JM5ZFJ5VHYU2V.json","view_paper":"https://pith.science/paper/BENDYJ7O","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2310.05939&json=true","fetch_graph":"https://pith.science/api/pith-number/BENDYJ7OEFKW7JM5ZFJ5VHYU2V/graph.json","fetch_events":"https://pith.science/api/pith-number/BENDYJ7OEFKW7JM5ZFJ5VHYU2V/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/BENDYJ7OEFKW7JM5ZFJ5VHYU2V/action/timestamp_anchor","attest_storage":"https://pith.science/pith/BENDYJ7OEFKW7JM5ZFJ5VHYU2V/action/storage_attestation","attest_author":"https://pith.science/pith/BENDYJ7OEFKW7JM5ZFJ5VHYU2V/action/author_attestation","sign_citation":"https://pith.science/pith/BENDYJ7OEFKW7JM5ZFJ5VHYU2V/action/citation_signature","submit_replication":"https://pith.science/pith/BENDYJ7OEFKW7JM5ZFJ5VHYU2V/action/replication_record"}},"created_at":"2026-07-05T06:59:02.096733+00:00","updated_at":"2026-07-05T06:59:02.096733+00:00"}