{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:PKLU3CGSX3NUMKQL7WBOAUKQHH","short_pith_number":"pith:PKLU3CGS","schema_version":"1.0","canonical_sha256":"7a974d88d2bedb462a0bfd82e0515039d8e032190d3579ce2477174428d4e4c2","source":{"kind":"arxiv","id":"2504.10112","version":2},"attestation_state":"computed","paper":{"title":"Benchmarking Practices in LLM-driven Offensive Security: Testbeds, Metrics, and Experiment Design","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CR","authors_text":"Andreas Happe, J\\\"urgen Cito","submitted_at":"2025-04-14T11:21:33Z","abstract_excerpt":"Large Language Models (LLMs) have emerged as a powerful approach for driving offensive penetration-testing tooling. Due to the opaque nature of LLMs, empirical methods are typically used to analyze their efficacy. The quality of this analysis is highly dependent on the chosen testbed, captured metrics and analysis methods employed.\n  This paper analyzes the methodology and benchmarking practices used for evaluating Large Language Model (LLM)-driven attacks, focusing on offensive uses of LLMs in cybersecurity. We review 19 research papers detailing 18 prototypes and their respective testbeds.\n "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2504.10112","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CR","submitted_at":"2025-04-14T11:21:33Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"46d4dbbbb6bc3de62b0c4cb25453421aa7a9aee6a0b65449c0353750442cb4e0","abstract_canon_sha256":"8af119748523e15734d63674bf224846e18b0e71bbdfcc3cf8a59ee1a8728901"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:22:18.255964Z","signature_b64":"EmPogt7KvzjsW7IkMrAuL6yt8G5IBg8sdJp7TgN81RGgKVRwnI9ZjhvIiOXZg7GTDmbfASyN348W8pSGGmYsAQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"7a974d88d2bedb462a0bfd82e0515039d8e032190d3579ce2477174428d4e4c2","last_reissued_at":"2026-07-05T11:22:18.255377Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:22:18.255377Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Benchmarking Practices in LLM-driven Offensive Security: Testbeds, Metrics, and Experiment Design","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CR","authors_text":"Andreas Happe, J\\\"urgen Cito","submitted_at":"2025-04-14T11:21:33Z","abstract_excerpt":"Large Language Models (LLMs) have emerged as a powerful approach for driving offensive penetration-testing tooling. Due to the opaque nature of LLMs, empirical methods are typically used to analyze their efficacy. The quality of this analysis is highly dependent on the chosen testbed, captured metrics and analysis methods employed.\n  This paper analyzes the methodology and benchmarking practices used for evaluating Large Language Model (LLM)-driven attacks, focusing on offensive uses of LLMs in cybersecurity. We review 19 research papers detailing 18 prototypes and their respective testbeds.\n "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2504.10112","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2504.10112/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2504.10112","created_at":"2026-07-05T11:22:18.255447+00:00"},{"alias_kind":"arxiv_version","alias_value":"2504.10112v2","created_at":"2026-07-05T11:22:18.255447+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2504.10112","created_at":"2026-07-05T11:22:18.255447+00:00"},{"alias_kind":"pith_short_12","alias_value":"PKLU3CGSX3NU","created_at":"2026-07-05T11:22:18.255447+00:00"},{"alias_kind":"pith_short_16","alias_value":"PKLU3CGSX3NUMKQL","created_at":"2026-07-05T11:22:18.255447+00:00"},{"alias_kind":"pith_short_8","alias_value":"PKLU3CGS","created_at":"2026-07-05T11:22:18.255447+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.25332","citing_title":"Decoupling Reconnaissance and Exploitation: Measuring the Capability Boundaries of LLM-Based Web Penetration Testing","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2606.05567","citing_title":"ZERO-APT: A Closed-Loop Adversarial Framework for LLM-Driven Automated Penetration Testing under Intelligent Defense","ref_index":56,"is_internal_anchor":false},{"citing_arxiv_id":"2606.29981","citing_title":"Hephaestus: Toward a Cybersecurity AI Scientist","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2605.30096","citing_title":"How Reliable Are AI Attackers Against a Fixed Vulnerable Target? A 400-Run Empirical Study of LLM Penetration Testing Consistency","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10834","citing_title":"From Controlled to the Wild: Evaluation of Pentesting Agents for the Real-World","ref_index":11,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/PKLU3CGSX3NUMKQL7WBOAUKQHH","json":"https://pith.science/pith/PKLU3CGSX3NUMKQL7WBOAUKQHH.json","graph_json":"https://pith.science/api/pith-number/PKLU3CGSX3NUMKQL7WBOAUKQHH/graph.json","events_json":"https://pith.science/api/pith-number/PKLU3CGSX3NUMKQL7WBOAUKQHH/events.json","paper":"https://pith.science/paper/PKLU3CGS"},"agent_actions":{"view_html":"https://pith.science/pith/PKLU3CGSX3NUMKQL7WBOAUKQHH","download_json":"https://pith.science/pith/PKLU3CGSX3NUMKQL7WBOAUKQHH.json","view_paper":"https://pith.science/paper/PKLU3CGS","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2504.10112&json=true","fetch_graph":"https://pith.science/api/pith-number/PKLU3CGSX3NUMKQL7WBOAUKQHH/graph.json","fetch_events":"https://pith.science/api/pith-number/PKLU3CGSX3NUMKQL7WBOAUKQHH/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/PKLU3CGSX3NUMKQL7WBOAUKQHH/action/timestamp_anchor","attest_storage":"https://pith.science/pith/PKLU3CGSX3NUMKQL7WBOAUKQHH/action/storage_attestation","attest_author":"https://pith.science/pith/PKLU3CGSX3NUMKQL7WBOAUKQHH/action/author_attestation","sign_citation":"https://pith.science/pith/PKLU3CGSX3NUMKQL7WBOAUKQHH/action/citation_signature","submit_replication":"https://pith.science/pith/PKLU3CGSX3NUMKQL7WBOAUKQHH/action/replication_record"}},"created_at":"2026-07-05T11:22:18.255447+00:00","updated_at":"2026-07-05T11:22:18.255447+00:00"}