{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:7Q6ET3R6ELPUNPJW2R4BN5TDOP","short_pith_number":"pith:7Q6ET3R6","schema_version":"1.0","canonical_sha256":"fc3c49ee3e22df46bd36d47816f66373e2ab2ec8afa0931211aa0062b240f533","source":{"kind":"arxiv","id":"2509.03331","version":1},"attestation_state":"computed","paper":{"title":"VulnRepairEval: An Exploit-Based Evaluation Framework for Assessing Large Language Model Vulnerability Repair Capabilities","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CR"],"primary_cat":"cs.SE","authors_text":"Bin Wu, Guangquan Xu, Jianfei Sun, Lingxiao Jiang, Qiang Hu, Wei Ma, Weizhe Wang, Yang Liu, Yao Zhang","submitted_at":"2025-09-03T14:06:10Z","abstract_excerpt":"The adoption of Large Language Models (LLMs) for automated software vulnerability patching has shown promising outcomes on carefully curated evaluation sets. Nevertheless, existing datasets predominantly rely on superficial validation methods rather than exploit-based verification, leading to overestimated performance in security-sensitive applications. This paper introduces VulnRepairEval, an evaluation framework anchored in functional Proof-of-Concept (PoC) exploits. Our framework delivers a comprehensive, containerized evaluation pipeline that enables reproducible differential assessment, w"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2509.03331","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.SE","submitted_at":"2025-09-03T14:06:10Z","cross_cats_sorted":["cs.CR"],"title_canon_sha256":"a81c8d74cbe1e36af5b3489a6b92718d0087bd1ffec56552769b2a894677ba91","abstract_canon_sha256":"68e6382ad5ce205dfd34e80bdd12f87e2d380ea71e6d232ad28e76309ae32897"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T12:04:20.603408Z","signature_b64":"uZMbXcrzpgX3i5oNHWFNgiq0Db0rTMfBamL9dw7Xv/m0StGZQxCc9Y1yDUypUetxSoe3sPj0s4Jcxnv9uldgBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"fc3c49ee3e22df46bd36d47816f66373e2ab2ec8afa0931211aa0062b240f533","last_reissued_at":"2026-07-05T12:04:20.602926Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T12:04:20.602926Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"VulnRepairEval: An Exploit-Based Evaluation Framework for Assessing Large Language Model Vulnerability Repair Capabilities","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CR"],"primary_cat":"cs.SE","authors_text":"Bin Wu, Guangquan Xu, Jianfei Sun, Lingxiao Jiang, Qiang Hu, Wei Ma, Weizhe Wang, Yang Liu, Yao Zhang","submitted_at":"2025-09-03T14:06:10Z","abstract_excerpt":"The adoption of Large Language Models (LLMs) for automated software vulnerability patching has shown promising outcomes on carefully curated evaluation sets. Nevertheless, existing datasets predominantly rely on superficial validation methods rather than exploit-based verification, leading to overestimated performance in security-sensitive applications. This paper introduces VulnRepairEval, an evaluation framework anchored in functional Proof-of-Concept (PoC) exploits. Our framework delivers a comprehensive, containerized evaluation pipeline that enables reproducible differential assessment, w"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2509.03331","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2509.03331/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2509.03331","created_at":"2026-07-05T12:04:20.602990+00:00"},{"alias_kind":"arxiv_version","alias_value":"2509.03331v1","created_at":"2026-07-05T12:04:20.602990+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2509.03331","created_at":"2026-07-05T12:04:20.602990+00:00"},{"alias_kind":"pith_short_12","alias_value":"7Q6ET3R6ELPU","created_at":"2026-07-05T12:04:20.602990+00:00"},{"alias_kind":"pith_short_16","alias_value":"7Q6ET3R6ELPUNPJW","created_at":"2026-07-05T12:04:20.602990+00:00"},{"alias_kind":"pith_short_8","alias_value":"7Q6ET3R6","created_at":"2026-07-05T12:04:20.602990+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.25973","citing_title":"Helpful or Harmful? Evaluating LLM-Assisted Vulnerability Patching via a Human Study","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2605.28893","citing_title":"Towards Demystifying and Repairing LLM-in-the-Loop Vulnerabilities","ref_index":53,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/7Q6ET3R6ELPUNPJW2R4BN5TDOP","json":"https://pith.science/pith/7Q6ET3R6ELPUNPJW2R4BN5TDOP.json","graph_json":"https://pith.science/api/pith-number/7Q6ET3R6ELPUNPJW2R4BN5TDOP/graph.json","events_json":"https://pith.science/api/pith-number/7Q6ET3R6ELPUNPJW2R4BN5TDOP/events.json","paper":"https://pith.science/paper/7Q6ET3R6"},"agent_actions":{"view_html":"https://pith.science/pith/7Q6ET3R6ELPUNPJW2R4BN5TDOP","download_json":"https://pith.science/pith/7Q6ET3R6ELPUNPJW2R4BN5TDOP.json","view_paper":"https://pith.science/paper/7Q6ET3R6","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2509.03331&json=true","fetch_graph":"https://pith.science/api/pith-number/7Q6ET3R6ELPUNPJW2R4BN5TDOP/graph.json","fetch_events":"https://pith.science/api/pith-number/7Q6ET3R6ELPUNPJW2R4BN5TDOP/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/7Q6ET3R6ELPUNPJW2R4BN5TDOP/action/timestamp_anchor","attest_storage":"https://pith.science/pith/7Q6ET3R6ELPUNPJW2R4BN5TDOP/action/storage_attestation","attest_author":"https://pith.science/pith/7Q6ET3R6ELPUNPJW2R4BN5TDOP/action/author_attestation","sign_citation":"https://pith.science/pith/7Q6ET3R6ELPUNPJW2R4BN5TDOP/action/citation_signature","submit_replication":"https://pith.science/pith/7Q6ET3R6ELPUNPJW2R4BN5TDOP/action/replication_record"}},"created_at":"2026-07-05T12:04:20.602990+00:00","updated_at":"2026-07-05T12:04:20.602990+00:00"}