{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2023:NOXHADKS2JIMLEP7BXT6IW6V67","short_pith_number":"pith:NOXHADKS","canonical_record":{"source":{"id":"2310.13565","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2023-10-20T15:04:42Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"8b20b3db326b605f02734c98a3c704abf259860734f782f5ea014c7477c3f8e7","abstract_canon_sha256":"3fd6b4d4d9eb784bb03c530eb64b23751aeca1c09785c5c8ac7e0ea0e76de45a"},"schema_version":"1.0"},"canonical_sha256":"6bae700d52d250c591ff0de7e45bd5f7c5c8052507db2d1f566b2cbc74c21f0d","source":{"kind":"arxiv","id":"2310.13565","version":1},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2310.13565","created_at":"2026-07-05T07:03:12Z"},{"alias_kind":"arxiv_version","alias_value":"2310.13565v1","created_at":"2026-07-05T07:03:12Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2310.13565","created_at":"2026-07-05T07:03:12Z"},{"alias_kind":"pith_short_12","alias_value":"NOXHADKS2JIM","created_at":"2026-07-05T07:03:12Z"},{"alias_kind":"pith_short_16","alias_value":"NOXHADKS2JIMLEP7","created_at":"2026-07-05T07:03:12Z"},{"alias_kind":"pith_short_8","alias_value":"NOXHADKS","created_at":"2026-07-05T07:03:12Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2023:NOXHADKS2JIMLEP7BXT6IW6V67","target":"record","payload":{"canonical_record":{"source":{"id":"2310.13565","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2023-10-20T15:04:42Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"8b20b3db326b605f02734c98a3c704abf259860734f782f5ea014c7477c3f8e7","abstract_canon_sha256":"3fd6b4d4d9eb784bb03c530eb64b23751aeca1c09785c5c8ac7e0ea0e76de45a"},"schema_version":"1.0"},"canonical_sha256":"6bae700d52d250c591ff0de7e45bd5f7c5c8052507db2d1f566b2cbc74c21f0d","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:03:12.458693Z","signature_b64":"TKJRHBiUIZhqc8KAVzHVCdpGzRuwAmKv7KjcNtj4xBz5ETswAHn4Tgvl41QDFcuv9n0EEFDlaBbV2BeE7taZAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"6bae700d52d250c591ff0de7e45bd5f7c5c8052507db2d1f566b2cbc74c21f0d","last_reissued_at":"2026-07-05T07:03:12.458212Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:03:12.458212Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2310.13565","source_version":1,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T07:03:12Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"GyBsai4Mr0J3hCDD4QNRVEPVz0oaMKxpoj5SZy5bA4ZuTZNPYxhK3qnOLcYtjhOBxpFNLsFZ8JY86dBLV/qkCA==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-01T01:40:23.922615Z"},"content_sha256":"d08535c524423f926ed8bdfacbae882d14f044e0334b24f53ad18aab98d8beec","schema_version":"1.0","event_id":"sha256:d08535c524423f926ed8bdfacbae882d14f044e0334b24f53ad18aab98d8beec"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2023:NOXHADKS2JIMLEP7BXT6IW6V67","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"Reward Shaping for Happier Autonomous Cyber Security Agents","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Chris Hicks, Elizabeth Bates, Vasilios Mavroudis","submitted_at":"2023-10-20T15:04:42Z","abstract_excerpt":"As machine learning models become more capable, they have exhibited increased potential in solving complex tasks. One of the most promising directions uses deep reinforcement learning to train autonomous agents in computer network defense tasks. This work studies the impact of the reward signal that is provided to the agents when training for this task. Due to the nature of cybersecurity tasks, the reward signal is typically 1) in the form of penalties (e.g., when a compromise occurs), and 2) distributed sparsely across each defense episode. Such reward characteristics are atypical of classic "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2310.13565","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2310.13565/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T07:03:12Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"coZQ3ud0Msh0dQshTTHDlDDufJiUTmUC4Knhnk/H1pOeRDW8rT2JJxpZX1GBt/y2KYY7woYAugDeMHZ9S4ZaCg==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-01T01:40:23.924281Z"},"content_sha256":"2b4bd52597b0dbffb19cba73b1cf91d0bcbb316bbdb1fc803a6781a56b5b42a3","schema_version":"1.0","event_id":"sha256:2b4bd52597b0dbffb19cba73b1cf91d0bcbb316bbdb1fc803a6781a56b5b42a3"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/NOXHADKS2JIMLEP7BXT6IW6V67/bundle.json","state_url":"https://pith.science/pith/NOXHADKS2JIMLEP7BXT6IW6V67/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/NOXHADKS2JIMLEP7BXT6IW6V67/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-01T01:40:23Z","links":{"resolver":"https://pith.science/pith/NOXHADKS2JIMLEP7BXT6IW6V67","bundle":"https://pith.science/pith/NOXHADKS2JIMLEP7BXT6IW6V67/bundle.json","state":"https://pith.science/pith/NOXHADKS2JIMLEP7BXT6IW6V67/state.json","well_known_bundle":"https://pith.science/.well-known/pith/NOXHADKS2JIMLEP7BXT6IW6V67/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2023:NOXHADKS2JIMLEP7BXT6IW6V67","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"3fd6b4d4d9eb784bb03c530eb64b23751aeca1c09785c5c8ac7e0ea0e76de45a","cross_cats_sorted":["cs.AI"],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2023-10-20T15:04:42Z","title_canon_sha256":"8b20b3db326b605f02734c98a3c704abf259860734f782f5ea014c7477c3f8e7"},"schema_version":"1.0","source":{"id":"2310.13565","kind":"arxiv","version":1}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2310.13565","created_at":"2026-07-05T07:03:12Z"},{"alias_kind":"arxiv_version","alias_value":"2310.13565v1","created_at":"2026-07-05T07:03:12Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2310.13565","created_at":"2026-07-05T07:03:12Z"},{"alias_kind":"pith_short_12","alias_value":"NOXHADKS2JIM","created_at":"2026-07-05T07:03:12Z"},{"alias_kind":"pith_short_16","alias_value":"NOXHADKS2JIMLEP7","created_at":"2026-07-05T07:03:12Z"},{"alias_kind":"pith_short_8","alias_value":"NOXHADKS","created_at":"2026-07-05T07:03:12Z"}],"graph_snapshots":[{"event_id":"sha256:2b4bd52597b0dbffb19cba73b1cf91d0bcbb316bbdb1fc803a6781a56b5b42a3","target":"graph","created_at":"2026-07-05T07:03:12Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2310.13565/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"As machine learning models become more capable, they have exhibited increased potential in solving complex tasks. One of the most promising directions uses deep reinforcement learning to train autonomous agents in computer network defense tasks. This work studies the impact of the reward signal that is provided to the agents when training for this task. Due to the nature of cybersecurity tasks, the reward signal is typically 1) in the form of penalties (e.g., when a compromise occurs), and 2) distributed sparsely across each defense episode. Such reward characteristics are atypical of classic ","authors_text":"Chris Hicks, Elizabeth Bates, Vasilios Mavroudis","cross_cats":["cs.AI"],"headline":"","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2023-10-20T15:04:42Z","title":"Reward Shaping for Happier Autonomous Cyber Security Agents"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2310.13565","kind":"arxiv","version":1},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:d08535c524423f926ed8bdfacbae882d14f044e0334b24f53ad18aab98d8beec","target":"record","created_at":"2026-07-05T07:03:12Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"3fd6b4d4d9eb784bb03c530eb64b23751aeca1c09785c5c8ac7e0ea0e76de45a","cross_cats_sorted":["cs.AI"],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2023-10-20T15:04:42Z","title_canon_sha256":"8b20b3db326b605f02734c98a3c704abf259860734f782f5ea014c7477c3f8e7"},"schema_version":"1.0","source":{"id":"2310.13565","kind":"arxiv","version":1}},"canonical_sha256":"6bae700d52d250c591ff0de7e45bd5f7c5c8052507db2d1f566b2cbc74c21f0d","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"6bae700d52d250c591ff0de7e45bd5f7c5c8052507db2d1f566b2cbc74c21f0d","first_computed_at":"2026-07-05T07:03:12.458212Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T07:03:12.458212Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"TKJRHBiUIZhqc8KAVzHVCdpGzRuwAmKv7KjcNtj4xBz5ETswAHn4Tgvl41QDFcuv9n0EEFDlaBbV2BeE7taZAg==","signature_status":"signed_v1","signed_at":"2026-07-05T07:03:12.458693Z","signed_message":"canonical_sha256_bytes"},"source_id":"2310.13565","source_kind":"arxiv","source_version":1}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:d08535c524423f926ed8bdfacbae882d14f044e0334b24f53ad18aab98d8beec","sha256:2b4bd52597b0dbffb19cba73b1cf91d0bcbb316bbdb1fc803a6781a56b5b42a3"],"state_sha256":"70517ccb6966c0d56ca9094c7de9eaf3b50132fbb5eb692ca5ede352ef742b8e"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"NPV8S9RE1x7tK99nQKZV3KWNHO736rn/troeJl/lsS/v7VGsfPBB7XF7YQaZBY6lr0GK0u6v3i7urlgOkBqAAQ==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-01T01:40:23.930286Z","bundle_sha256":"5a286802b0bc361886485270f57fe083944c68353be9c6ff7a1c721ed70787e0"}}