{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:P5UPY46Q3G2XQ3RUWAVEH5XYY7","short_pith_number":"pith:P5UPY46Q","schema_version":"1.0","canonical_sha256":"7f68fc73d0d9b5786e34b02a43f6f8c7f481b36bdb6cece60e1740d09306f7ca","source":{"kind":"arxiv","id":"2410.17141","version":4},"attestation_state":"computed","paper":{"title":"Towards Automated Penetration Testing: Introducing LLM Benchmark, Analysis, and Improvements","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CR","authors_text":"Edward Kim, Isamu Isozaki, Manil Shrestha, Rick Console","submitted_at":"2024-10-22T16:18:41Z","abstract_excerpt":"Hacking poses a significant threat to cybersecurity, inflicting billions of dollars in damages annually. To mitigate these risks, ethical hacking, or penetration testing, is employed to identify vulnerabilities in systems and networks. Recent advancements in large language models (LLMs) have shown potential across various domains, including cybersecurity. However, there is currently no comprehensive, open, automated, end-to-end penetration testing benchmark to drive progress and evaluate the capabilities of these models in security contexts. This paper introduces a novel open benchmark for LLM"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.17141","kind":"arxiv","version":4},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CR","submitted_at":"2024-10-22T16:18:41Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"ad837f4f1af681b8e1054993be1078d1d73cebfbd4f8c8c5d015c1efe4dedc95","abstract_canon_sha256":"c257f1c4a1283718c1129841048e27fd65898c7119dda962c98ae6c51ce88366"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:17:50.976351Z","signature_b64":"SlR88TJc4Z8i5TZe1+UTeNatgIUjg4KvsyADcFRIkkWxRKpgc/SFOZdkADgpgWndAKFVP6LQASvrWMj8XcysDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"7f68fc73d0d9b5786e34b02a43f6f8c7f481b36bdb6cece60e1740d09306f7ca","last_reissued_at":"2026-07-05T10:17:50.975871Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:17:50.975871Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Towards Automated Penetration Testing: Introducing LLM Benchmark, Analysis, and Improvements","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CR","authors_text":"Edward Kim, Isamu Isozaki, Manil Shrestha, Rick Console","submitted_at":"2024-10-22T16:18:41Z","abstract_excerpt":"Hacking poses a significant threat to cybersecurity, inflicting billions of dollars in damages annually. To mitigate these risks, ethical hacking, or penetration testing, is employed to identify vulnerabilities in systems and networks. Recent advancements in large language models (LLMs) have shown potential across various domains, including cybersecurity. However, there is currently no comprehensive, open, automated, end-to-end penetration testing benchmark to drive progress and evaluate the capabilities of these models in security contexts. This paper introduces a novel open benchmark for LLM"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.17141","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.17141/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.17141","created_at":"2026-07-05T10:17:50.975928+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.17141v4","created_at":"2026-07-05T10:17:50.975928+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.17141","created_at":"2026-07-05T10:17:50.975928+00:00"},{"alias_kind":"pith_short_12","alias_value":"P5UPY46Q3G2X","created_at":"2026-07-05T10:17:50.975928+00:00"},{"alias_kind":"pith_short_16","alias_value":"P5UPY46Q3G2XQ3RU","created_at":"2026-07-05T10:17:50.975928+00:00"},{"alias_kind":"pith_short_8","alias_value":"P5UPY46Q","created_at":"2026-07-05T10:17:50.975928+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.30096","citing_title":"How Reliable Are AI Attackers Against a Fixed Vulnerable Target? A 400-Run Empirical Study of LLM Penetration Testing Consistency","ref_index":9,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/P5UPY46Q3G2XQ3RUWAVEH5XYY7","json":"https://pith.science/pith/P5UPY46Q3G2XQ3RUWAVEH5XYY7.json","graph_json":"https://pith.science/api/pith-number/P5UPY46Q3G2XQ3RUWAVEH5XYY7/graph.json","events_json":"https://pith.science/api/pith-number/P5UPY46Q3G2XQ3RUWAVEH5XYY7/events.json","paper":"https://pith.science/paper/P5UPY46Q"},"agent_actions":{"view_html":"https://pith.science/pith/P5UPY46Q3G2XQ3RUWAVEH5XYY7","download_json":"https://pith.science/pith/P5UPY46Q3G2XQ3RUWAVEH5XYY7.json","view_paper":"https://pith.science/paper/P5UPY46Q","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.17141&json=true","fetch_graph":"https://pith.science/api/pith-number/P5UPY46Q3G2XQ3RUWAVEH5XYY7/graph.json","fetch_events":"https://pith.science/api/pith-number/P5UPY46Q3G2XQ3RUWAVEH5XYY7/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/P5UPY46Q3G2XQ3RUWAVEH5XYY7/action/timestamp_anchor","attest_storage":"https://pith.science/pith/P5UPY46Q3G2XQ3RUWAVEH5XYY7/action/storage_attestation","attest_author":"https://pith.science/pith/P5UPY46Q3G2XQ3RUWAVEH5XYY7/action/author_attestation","sign_citation":"https://pith.science/pith/P5UPY46Q3G2XQ3RUWAVEH5XYY7/action/citation_signature","submit_replication":"https://pith.science/pith/P5UPY46Q3G2XQ3RUWAVEH5XYY7/action/replication_record"}},"created_at":"2026-07-05T10:17:50.975928+00:00","updated_at":"2026-07-05T10:17:50.975928+00:00"}