{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:JLK4YQWGKHXUZRDJCSILALA5OT","short_pith_number":"pith:JLK4YQWG","schema_version":"1.0","canonical_sha256":"4ad5cc42c651ef4cc4691490b02c1d74c57f6ac28309705e95eee5422ffc1fc4","source":{"kind":"arxiv","id":"2503.17332","version":4},"attestation_state":"computed","paper":{"title":"CVE-Bench: A Benchmark for AI Agents' Ability to Exploit Real-World Web Application Vulnerabilities","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CR","authors_text":"Adarsh Danda, Akul Gupta, Antony Kellermann, Avi Dhir, Conner Jensen, Daniel Kang, Dylan Bowman, Eric Ihli, Jason Benn, Jet Geronimo, Kaicheng Yu, Philip Li, Richard Fang, Sudhit Rao, Twm Stone, Yuxuan Zhu","submitted_at":"2025-03-21T17:32:32Z","abstract_excerpt":"Large language model (LLM) agents are increasingly capable of autonomously conducting cyberattacks, posing significant threats to existing applications. This growing risk highlights the urgent need for a real-world benchmark to evaluate the ability of LLM agents to exploit web application vulnerabilities. However, existing benchmarks fall short as they are limited to abstracted Capture the Flag competitions or lack comprehensive coverage. Building a benchmark for real-world vulnerabilities involves both specialized expertise to reproduce exploits and a systematic approach to evaluating unpredi"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2503.17332","kind":"arxiv","version":4},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CR","submitted_at":"2025-03-21T17:32:32Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"75333c9ac78d0bd6e2dcbcfcd1135fbbf6218be6d535c112c598c112170963fc","abstract_canon_sha256":"84a1c1f36e789035a99988a367b7fda9f3af4a54ba1d78b164e9ba91942992aa"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:26:03.586521Z","signature_b64":"yWT7wax2sqY5NOIU/fc3IlFBw+q1f/bL0ghJ91TsgKmIUIm46X2+LqnobYvS5MaKN7a//ZjXcLIRFaEaILkcBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"4ad5cc42c651ef4cc4691490b02c1d74c57f6ac28309705e95eee5422ffc1fc4","last_reissued_at":"2026-07-05T11:26:03.586023Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:26:03.586023Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"CVE-Bench: A Benchmark for AI Agents' Ability to Exploit Real-World Web Application Vulnerabilities","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CR","authors_text":"Adarsh Danda, Akul Gupta, Antony Kellermann, Avi Dhir, Conner Jensen, Daniel Kang, Dylan Bowman, Eric Ihli, Jason Benn, Jet Geronimo, Kaicheng Yu, Philip Li, Richard Fang, Sudhit Rao, Twm Stone, Yuxuan Zhu","submitted_at":"2025-03-21T17:32:32Z","abstract_excerpt":"Large language model (LLM) agents are increasingly capable of autonomously conducting cyberattacks, posing significant threats to existing applications. This growing risk highlights the urgent need for a real-world benchmark to evaluate the ability of LLM agents to exploit web application vulnerabilities. However, existing benchmarks fall short as they are limited to abstracted Capture the Flag competitions or lack comprehensive coverage. Building a benchmark for real-world vulnerabilities involves both specialized expertise to reproduce exploits and a systematic approach to evaluating unpredi"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2503.17332","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2503.17332/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2503.17332","created_at":"2026-07-05T11:26:03.586082+00:00"},{"alias_kind":"arxiv_version","alias_value":"2503.17332v4","created_at":"2026-07-05T11:26:03.586082+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2503.17332","created_at":"2026-07-05T11:26:03.586082+00:00"},{"alias_kind":"pith_short_12","alias_value":"JLK4YQWGKHXU","created_at":"2026-07-05T11:26:03.586082+00:00"},{"alias_kind":"pith_short_16","alias_value":"JLK4YQWGKHXUZRDJ","created_at":"2026-07-05T11:26:03.586082+00:00"},{"alias_kind":"pith_short_8","alias_value":"JLK4YQWG","created_at":"2026-07-05T11:26:03.586082+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":11,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.07109","citing_title":"Certifying Ghosts: How Cybersecurity AI Agents Break the EU Cyber Resilience Act","ref_index":24,"is_internal_anchor":true},{"citing_arxiv_id":"2606.14295","citing_title":"AgentCyberRange: Benchmarking Frontier AI Systems in Realistic Cyber Ranges","ref_index":48,"is_internal_anchor":false},{"citing_arxiv_id":"2606.13079","citing_title":"The Emergence of Autonomous Penetration Capabilities in Large Language Model-Powered AI Systems","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2606.05567","citing_title":"ZERO-APT: A Closed-Loop Adversarial Framework for LLM-Driven Automated Penetration Testing under Intelligent Defense","ref_index":64,"is_internal_anchor":false},{"citing_arxiv_id":"2606.03453","citing_title":"FORGE: Multi-Agent Graduated Exploitation and Detection Engineering","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2606.13079","citing_title":"The Emergence of Autonomous Penetration Capabilities in Large Language Model-Powered AI Systems","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2605.26195","citing_title":"CyberEvolver: Structured Self-Evolution for Cybersecurity Agents On the Fly","ref_index":82,"is_internal_anchor":false},{"citing_arxiv_id":"2509.13021","citing_title":"xOffense: An Autonomous Multi-Agent Framework for Penetration Testing with Domain-Adapted Large Language Models","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10834","citing_title":"From Controlled to the Wild: Evaluation of Pentesting Agents for the Real-World","ref_index":51,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07830","citing_title":"CyBiasBench: Benchmarking Bias in LLM Agents for Cyber-Attack Scenarios","ref_index":36,"is_internal_anchor":false},{"citing_arxiv_id":"2604.13764","citing_title":"RealVuln: Benchmarking Rule-Based, General-Purpose LLM, and Security-Specialized Scanners on Real-World Code","ref_index":26,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/JLK4YQWGKHXUZRDJCSILALA5OT","json":"https://pith.science/pith/JLK4YQWGKHXUZRDJCSILALA5OT.json","graph_json":"https://pith.science/api/pith-number/JLK4YQWGKHXUZRDJCSILALA5OT/graph.json","events_json":"https://pith.science/api/pith-number/JLK4YQWGKHXUZRDJCSILALA5OT/events.json","paper":"https://pith.science/paper/JLK4YQWG"},"agent_actions":{"view_html":"https://pith.science/pith/JLK4YQWGKHXUZRDJCSILALA5OT","download_json":"https://pith.science/pith/JLK4YQWGKHXUZRDJCSILALA5OT.json","view_paper":"https://pith.science/paper/JLK4YQWG","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2503.17332&json=true","fetch_graph":"https://pith.science/api/pith-number/JLK4YQWGKHXUZRDJCSILALA5OT/graph.json","fetch_events":"https://pith.science/api/pith-number/JLK4YQWGKHXUZRDJCSILALA5OT/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/JLK4YQWGKHXUZRDJCSILALA5OT/action/timestamp_anchor","attest_storage":"https://pith.science/pith/JLK4YQWGKHXUZRDJCSILALA5OT/action/storage_attestation","attest_author":"https://pith.science/pith/JLK4YQWGKHXUZRDJCSILALA5OT/action/author_attestation","sign_citation":"https://pith.science/pith/JLK4YQWGKHXUZRDJCSILALA5OT/action/citation_signature","submit_replication":"https://pith.science/pith/JLK4YQWGKHXUZRDJCSILALA5OT/action/replication_record"}},"created_at":"2026-07-05T11:26:03.586082+00:00","updated_at":"2026-07-05T11:26:03.586082+00:00"}