{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:OMHJLO544X7ZMY4OU6X4EDLHDF","short_pith_number":"pith:OMHJLO54","schema_version":"1.0","canonical_sha256":"730e95bbbce5ff96638ea7afc20d6719487385905ed8bf2fecba5a24bf150023","source":{"kind":"arxiv","id":"2509.20008","version":2},"attestation_state":"computed","paper":{"title":"Learning Robust Penetration Testing Policies under Partial Observability: A systematic evaluation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CR"],"primary_cat":"cs.LG","authors_text":"Pieter Libin, Raphael Simon, Wim Mees","submitted_at":"2025-09-24T11:27:54Z","abstract_excerpt":"Penetration testing, the simulation of cyberattacks to identify security vulnerabilities, presents a sequential decision-making problem well-suited for reinforcement learning (RL) automation. Like many applications of RL to real-world problems, partial observability presents a major challenge, as it invalidates the Markov property present in Markov Decision Processes (MDPs). Partially Observable MDPs require history aggregation or belief state estimation to learn successful policies. We investigate stochastic, partially observable penetration testing scenarios over host networks of varying siz"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2509.20008","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-09-24T11:27:54Z","cross_cats_sorted":["cs.CR"],"title_canon_sha256":"c64a3f6aeec868b081709c90a8221b8cae2950e7ed8dfdd9cd6137a13525e646","abstract_canon_sha256":"1ee8cd09d19d302d81a11df221a527d18de47cdbb6de004d73a5c2dc36620984"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-06-26T01:15:46.043485Z","signature_b64":"lyfMGDT3TCVOYFaK0gAiijzzibH9W4LkMtj71HZT7UJ2VBq8+2vQgckd8i04qQEvugmu8qEAhVOecv/tyEXxAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"730e95bbbce5ff96638ea7afc20d6719487385905ed8bf2fecba5a24bf150023","last_reissued_at":"2026-06-26T01:15:46.042980Z","signature_status":"signed_v1","first_computed_at":"2026-06-26T01:15:46.042980Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Learning Robust Penetration Testing Policies under Partial Observability: A systematic evaluation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CR"],"primary_cat":"cs.LG","authors_text":"Pieter Libin, Raphael Simon, Wim Mees","submitted_at":"2025-09-24T11:27:54Z","abstract_excerpt":"Penetration testing, the simulation of cyberattacks to identify security vulnerabilities, presents a sequential decision-making problem well-suited for reinforcement learning (RL) automation. Like many applications of RL to real-world problems, partial observability presents a major challenge, as it invalidates the Markov property present in Markov Decision Processes (MDPs). Partially Observable MDPs require history aggregation or belief state estimation to learn successful policies. We investigate stochastic, partially observable penetration testing scenarios over host networks of varying siz"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2509.20008","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2509.20008/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2509.20008","created_at":"2026-06-26T01:15:46.043040+00:00"},{"alias_kind":"arxiv_version","alias_value":"2509.20008v2","created_at":"2026-06-26T01:15:46.043040+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2509.20008","created_at":"2026-06-26T01:15:46.043040+00:00"},{"alias_kind":"pith_short_12","alias_value":"OMHJLO544X7Z","created_at":"2026-06-26T01:15:46.043040+00:00"},{"alias_kind":"pith_short_16","alias_value":"OMHJLO544X7ZMY4O","created_at":"2026-06-26T01:15:46.043040+00:00"},{"alias_kind":"pith_short_8","alias_value":"OMHJLO54","created_at":"2026-06-26T01:15:46.043040+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/OMHJLO544X7ZMY4OU6X4EDLHDF","json":"https://pith.science/pith/OMHJLO544X7ZMY4OU6X4EDLHDF.json","graph_json":"https://pith.science/api/pith-number/OMHJLO544X7ZMY4OU6X4EDLHDF/graph.json","events_json":"https://pith.science/api/pith-number/OMHJLO544X7ZMY4OU6X4EDLHDF/events.json","paper":"https://pith.science/paper/OMHJLO54"},"agent_actions":{"view_html":"https://pith.science/pith/OMHJLO544X7ZMY4OU6X4EDLHDF","download_json":"https://pith.science/pith/OMHJLO544X7ZMY4OU6X4EDLHDF.json","view_paper":"https://pith.science/paper/OMHJLO54","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2509.20008&json=true","fetch_graph":"https://pith.science/api/pith-number/OMHJLO544X7ZMY4OU6X4EDLHDF/graph.json","fetch_events":"https://pith.science/api/pith-number/OMHJLO544X7ZMY4OU6X4EDLHDF/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/OMHJLO544X7ZMY4OU6X4EDLHDF/action/timestamp_anchor","attest_storage":"https://pith.science/pith/OMHJLO544X7ZMY4OU6X4EDLHDF/action/storage_attestation","attest_author":"https://pith.science/pith/OMHJLO544X7ZMY4OU6X4EDLHDF/action/author_attestation","sign_citation":"https://pith.science/pith/OMHJLO544X7ZMY4OU6X4EDLHDF/action/citation_signature","submit_replication":"https://pith.science/pith/OMHJLO544X7ZMY4OU6X4EDLHDF/action/replication_record"}},"created_at":"2026-06-26T01:15:46.043040+00:00","updated_at":"2026-06-26T01:15:46.043040+00:00"}