{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:GT2GC4BNUZ3VJCWLBAMRW7SJS5","short_pith_number":"pith:GT2GC4BN","schema_version":"1.0","canonical_sha256":"34f461702da677548acb08191b7e499769069ff9884ca8110a7542fce70d09ba","source":{"kind":"arxiv","id":"2406.05590","version":3},"attestation_state":"computed","paper":{"title":"NYU CTF Bench: A Scalable Open-Source Benchmark Dataset for Evaluating LLMs in Offensive Security","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CY","cs.LG"],"primary_cat":"cs.CR","authors_text":"Boyuan Chen, Brendan Dolan-Gavitt, Farshad Khorrami, Haoran Xi, Kimberly Milner, Max Yin, Meet Udeshi, Minghao Shao, Muhammad Shafique, Prashanth Krishnamurthy, Ramesh Karri, Siddharth Garg, Sofija Jancheska","submitted_at":"2024-06-08T22:21:42Z","abstract_excerpt":"Large Language Models (LLMs) are being deployed across various domains today. However, their capacity to solve Capture the Flag (CTF) challenges in cybersecurity has not been thoroughly evaluated. To address this, we develop a novel method to assess LLMs in solving CTF challenges by creating a scalable, open-source benchmark database specifically designed for these applications. This database includes metadata for LLM testing and adaptive learning, compiling a diverse range of CTF challenges from popular competitions. Utilizing the advanced function calling capabilities of LLMs, we build a ful"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2406.05590","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CR","submitted_at":"2024-06-08T22:21:42Z","cross_cats_sorted":["cs.AI","cs.CY","cs.LG"],"title_canon_sha256":"3a3fe51f63959ddead9a1e3edf16a9b6ddcf446c88f9ecafabe609c39929c8ee","abstract_canon_sha256":"3eb6c838a7f139fa5cac7582ab28b01de23441b2a54de26e16628665437f1242"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:16:10.412476Z","signature_b64":"htFtNCwEBS8PkTaAV4D6+I1W3FDKixdR0a1nKzsF1bpDa0XEsxbVOBgKmoVNLhGHtDzIU/Mq6nAQA0V7bEuvDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"34f461702da677548acb08191b7e499769069ff9884ca8110a7542fce70d09ba","last_reissued_at":"2026-07-05T10:16:10.411906Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:16:10.411906Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"NYU CTF Bench: A Scalable Open-Source Benchmark Dataset for Evaluating LLMs in Offensive Security","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CY","cs.LG"],"primary_cat":"cs.CR","authors_text":"Boyuan Chen, Brendan Dolan-Gavitt, Farshad Khorrami, Haoran Xi, Kimberly Milner, Max Yin, Meet Udeshi, Minghao Shao, Muhammad Shafique, Prashanth Krishnamurthy, Ramesh Karri, Siddharth Garg, Sofija Jancheska","submitted_at":"2024-06-08T22:21:42Z","abstract_excerpt":"Large Language Models (LLMs) are being deployed across various domains today. However, their capacity to solve Capture the Flag (CTF) challenges in cybersecurity has not been thoroughly evaluated. To address this, we develop a novel method to assess LLMs in solving CTF challenges by creating a scalable, open-source benchmark database specifically designed for these applications. This database includes metadata for LLM testing and adaptive learning, compiling a diverse range of CTF challenges from popular competitions. Utilizing the advanced function calling capabilities of LLMs, we build a ful"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.05590","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.05590/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2406.05590","created_at":"2026-07-05T10:16:10.411980+00:00"},{"alias_kind":"arxiv_version","alias_value":"2406.05590v3","created_at":"2026-07-05T10:16:10.411980+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.05590","created_at":"2026-07-05T10:16:10.411980+00:00"},{"alias_kind":"pith_short_12","alias_value":"GT2GC4BNUZ3V","created_at":"2026-07-05T10:16:10.411980+00:00"},{"alias_kind":"pith_short_16","alias_value":"GT2GC4BNUZ3VJCWL","created_at":"2026-07-05T10:16:10.411980+00:00"},{"alias_kind":"pith_short_8","alias_value":"GT2GC4BN","created_at":"2026-07-05T10:16:10.411980+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":9,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2607.01764","citing_title":"Mastermind: Strategy-grounded Learning for Repository-Scale Vulnerability Reproduction","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2605.23243","citing_title":"Are Frontier LLMs Ready for Cybersecurity? Evidence for Vertical Foundation Models from Dual-Mode Vulnerability Benchmarks","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2605.26195","citing_title":"CyberEvolver: Structured Self-Evolution for Cybersecurity Agents On the Fly","ref_index":54,"is_internal_anchor":false},{"citing_arxiv_id":"2605.29115","citing_title":"unix-ctf: Procedural Environments for Unix-Competence Reinforcement Learning","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2605.23243","citing_title":"Are Frontier LLMs Ready for Cybersecurity? Evidence for Vertical Foundation Models from Dual-Mode Vulnerability Benchmarks","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2604.24184","citing_title":"Dynamic Cyber Ranges","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2604.19533","citing_title":"Cyber Defense Benchmark: Agentic Threat Hunting Evaluation for LLMs in SecOps","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2604.17159","citing_title":"Systematic Capability Benchmarking of Frontier Large Language Models for Offensive Cyber Tasks","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2604.20801","citing_title":"Synthesizing Multi-Agent Harnesses for Vulnerability Discovery","ref_index":35,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/GT2GC4BNUZ3VJCWLBAMRW7SJS5","json":"https://pith.science/pith/GT2GC4BNUZ3VJCWLBAMRW7SJS5.json","graph_json":"https://pith.science/api/pith-number/GT2GC4BNUZ3VJCWLBAMRW7SJS5/graph.json","events_json":"https://pith.science/api/pith-number/GT2GC4BNUZ3VJCWLBAMRW7SJS5/events.json","paper":"https://pith.science/paper/GT2GC4BN"},"agent_actions":{"view_html":"https://pith.science/pith/GT2GC4BNUZ3VJCWLBAMRW7SJS5","download_json":"https://pith.science/pith/GT2GC4BNUZ3VJCWLBAMRW7SJS5.json","view_paper":"https://pith.science/paper/GT2GC4BN","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2406.05590&json=true","fetch_graph":"https://pith.science/api/pith-number/GT2GC4BNUZ3VJCWLBAMRW7SJS5/graph.json","fetch_events":"https://pith.science/api/pith-number/GT2GC4BNUZ3VJCWLBAMRW7SJS5/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/GT2GC4BNUZ3VJCWLBAMRW7SJS5/action/timestamp_anchor","attest_storage":"https://pith.science/pith/GT2GC4BNUZ3VJCWLBAMRW7SJS5/action/storage_attestation","attest_author":"https://pith.science/pith/GT2GC4BNUZ3VJCWLBAMRW7SJS5/action/author_attestation","sign_citation":"https://pith.science/pith/GT2GC4BNUZ3VJCWLBAMRW7SJS5/action/citation_signature","submit_replication":"https://pith.science/pith/GT2GC4BNUZ3VJCWLBAMRW7SJS5/action/replication_record"}},"created_at":"2026-07-05T10:16:10.411980+00:00","updated_at":"2026-07-05T10:16:10.411980+00:00"}