{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:YCQQBPHJEYUMDN6N47M5SMPNB4","short_pith_number":"pith:YCQQBPHJ","schema_version":"1.0","canonical_sha256":"c0a100bce92628c1b7cde7d9d931ed0f17d333be506921df4427ff1554105fa6","source":{"kind":"arxiv","id":"2402.11814","version":1},"attestation_state":"computed","paper":{"title":"An Empirical Evaluation of LLMs for Solving Offensive Security Challenges","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CR","authors_text":"Boyuan Chen, Brendan Dolan-Gavitt, Minghao Shao, Muhammad Shafique, Ramesh Karri, Siddharth Garg, Sofija Jancheska","submitted_at":"2024-02-19T04:08:44Z","abstract_excerpt":"Capture The Flag (CTF) challenges are puzzles related to computer security scenarios. With the advent of large language models (LLMs), more and more CTF participants are using LLMs to understand and solve the challenges. However, so far no work has evaluated the effectiveness of LLMs in solving CTF challenges with a fully automated workflow. We develop two CTF-solving workflows, human-in-the-loop (HITL) and fully-automated, to examine the LLMs' ability to solve a selected set of CTF challenges, prompted with information about the question. We collect human contestants' results on the same set "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2402.11814","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CR","submitted_at":"2024-02-19T04:08:44Z","cross_cats_sorted":[],"title_canon_sha256":"01a5c1796d634c1a6078854032596220788c805cd5f0fd4729d63ab61b6accf5","abstract_canon_sha256":"fae443b5339bc8e0f8fd26023ca202fd9a2a919eca2257c4a38321bb0caf6874"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:46:44.768678Z","signature_b64":"qWsdvweaYxE31w+KkPnBaHjoJrlEQuijdfWCRYFcLAcZqqdwfBSC2aZjNFe6azCZqPWezWE04KELohpwCeMAAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"c0a100bce92628c1b7cde7d9d931ed0f17d333be506921df4427ff1554105fa6","last_reissued_at":"2026-07-05T07:46:44.768265Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:46:44.768265Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"An Empirical Evaluation of LLMs for Solving Offensive Security Challenges","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CR","authors_text":"Boyuan Chen, Brendan Dolan-Gavitt, Minghao Shao, Muhammad Shafique, Ramesh Karri, Siddharth Garg, Sofija Jancheska","submitted_at":"2024-02-19T04:08:44Z","abstract_excerpt":"Capture The Flag (CTF) challenges are puzzles related to computer security scenarios. With the advent of large language models (LLMs), more and more CTF participants are using LLMs to understand and solve the challenges. However, so far no work has evaluated the effectiveness of LLMs in solving CTF challenges with a fully automated workflow. We develop two CTF-solving workflows, human-in-the-loop (HITL) and fully-automated, to examine the LLMs' ability to solve a selected set of CTF challenges, prompted with information about the question. We collect human contestants' results on the same set "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2402.11814","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2402.11814/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2402.11814","created_at":"2026-07-05T07:46:44.768322+00:00"},{"alias_kind":"arxiv_version","alias_value":"2402.11814v1","created_at":"2026-07-05T07:46:44.768322+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2402.11814","created_at":"2026-07-05T07:46:44.768322+00:00"},{"alias_kind":"pith_short_12","alias_value":"YCQQBPHJEYUM","created_at":"2026-07-05T07:46:44.768322+00:00"},{"alias_kind":"pith_short_16","alias_value":"YCQQBPHJEYUMDN6N","created_at":"2026-07-05T07:46:44.768322+00:00"},{"alias_kind":"pith_short_8","alias_value":"YCQQBPHJ","created_at":"2026-07-05T07:46:44.768322+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.06910","citing_title":"Benchmarking Large Language Models for IoC Recovery under Adversarial Code Obfuscation and Encryption","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2604.05719","citing_title":"Hackers or Hallucinators? A Comprehensive Analysis of LLM-Based Automated Penetration Testing","ref_index":97,"is_internal_anchor":false},{"citing_arxiv_id":"2604.06019","citing_title":"CritBench: A Framework for Evaluating Cybersecurity Capabilities of Large Language Models in IEC 61850 Digital Substation Environments","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2604.17948","citing_title":"RAVEN: Retrieval-Augmented Vulnerability Exploration Network for Memory Corruption Analysis in User Code and Binary Programs","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2604.17159","citing_title":"Systematic Capability Benchmarking of Frontier Large Language Models for Offensive Cyber Tasks","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2604.20389","citing_title":"CyberCertBench: Evaluating LLMs in Cybersecurity Certification Knowledge","ref_index":20,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/YCQQBPHJEYUMDN6N47M5SMPNB4","json":"https://pith.science/pith/YCQQBPHJEYUMDN6N47M5SMPNB4.json","graph_json":"https://pith.science/api/pith-number/YCQQBPHJEYUMDN6N47M5SMPNB4/graph.json","events_json":"https://pith.science/api/pith-number/YCQQBPHJEYUMDN6N47M5SMPNB4/events.json","paper":"https://pith.science/paper/YCQQBPHJ"},"agent_actions":{"view_html":"https://pith.science/pith/YCQQBPHJEYUMDN6N47M5SMPNB4","download_json":"https://pith.science/pith/YCQQBPHJEYUMDN6N47M5SMPNB4.json","view_paper":"https://pith.science/paper/YCQQBPHJ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2402.11814&json=true","fetch_graph":"https://pith.science/api/pith-number/YCQQBPHJEYUMDN6N47M5SMPNB4/graph.json","fetch_events":"https://pith.science/api/pith-number/YCQQBPHJEYUMDN6N47M5SMPNB4/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/YCQQBPHJEYUMDN6N47M5SMPNB4/action/timestamp_anchor","attest_storage":"https://pith.science/pith/YCQQBPHJEYUMDN6N47M5SMPNB4/action/storage_attestation","attest_author":"https://pith.science/pith/YCQQBPHJEYUMDN6N47M5SMPNB4/action/author_attestation","sign_citation":"https://pith.science/pith/YCQQBPHJEYUMDN6N47M5SMPNB4/action/citation_signature","submit_replication":"https://pith.science/pith/YCQQBPHJEYUMDN6N47M5SMPNB4/action/replication_record"}},"created_at":"2026-07-05T07:46:44.768322+00:00","updated_at":"2026-07-05T07:46:44.768322+00:00"}