{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:LHUY7QBLM3RRCXQGG2DIWUY3AE","short_pith_number":"pith:LHUY7QBL","schema_version":"1.0","canonical_sha256":"59e98fc02b66e3115e0636868b531b0124bc6915e8b9541a2a5ee782864adfa2","source":{"kind":"arxiv","id":"2311.12420","version":3},"attestation_state":"computed","paper":{"title":"How Far Have We Gone in Vulnerability Detection Using Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL","cs.CR"],"primary_cat":"cs.AI","authors_text":"Chao Zhang, Hao Wang, Wenyu Zhu, Yuchen Zhou, Zeyu Gao","submitted_at":"2023-11-21T08:20:39Z","abstract_excerpt":"As software becomes increasingly complex and prone to vulnerabilities, automated vulnerability detection is critically important, yet challenging. Given the significant successes of large language models (LLMs) in various tasks, there is growing anticipation of their efficacy in vulnerability detection. However, a quantitative understanding of their potential in vulnerability detection is still missing. To bridge this gap, we introduce a comprehensive vulnerability benchmark VulBench. This benchmark aggregates high-quality data from a wide range of CTF (Capture-the-Flag) challenges and real-wo"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2311.12420","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2023-11-21T08:20:39Z","cross_cats_sorted":["cs.CL","cs.CR"],"title_canon_sha256":"dc28945136ce7070423a6c7ecf1e58560974d64061bf73c511ac5d818a87fb06","abstract_canon_sha256":"bd28a13607958e43f0c27804fae1730f48ce36596300356b90f388fb359f6c66"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:27:09.494920Z","signature_b64":"OnZTCIpFBYtQHrCXba+/WCs5IXlYayH4bKm9DUnaKxTinf6fOVqTlcTJ3koxlP0PveK/mUzO2htrmVHPbGBZDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"59e98fc02b66e3115e0636868b531b0124bc6915e8b9541a2a5ee782864adfa2","last_reissued_at":"2026-07-05T07:27:09.494487Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:27:09.494487Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"How Far Have We Gone in Vulnerability Detection Using Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL","cs.CR"],"primary_cat":"cs.AI","authors_text":"Chao Zhang, Hao Wang, Wenyu Zhu, Yuchen Zhou, Zeyu Gao","submitted_at":"2023-11-21T08:20:39Z","abstract_excerpt":"As software becomes increasingly complex and prone to vulnerabilities, automated vulnerability detection is critically important, yet challenging. Given the significant successes of large language models (LLMs) in various tasks, there is growing anticipation of their efficacy in vulnerability detection. However, a quantitative understanding of their potential in vulnerability detection is still missing. To bridge this gap, we introduce a comprehensive vulnerability benchmark VulBench. This benchmark aggregates high-quality data from a wide range of CTF (Capture-the-Flag) challenges and real-wo"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2311.12420","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2311.12420/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2311.12420","created_at":"2026-07-05T07:27:09.494554+00:00"},{"alias_kind":"arxiv_version","alias_value":"2311.12420v3","created_at":"2026-07-05T07:27:09.494554+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2311.12420","created_at":"2026-07-05T07:27:09.494554+00:00"},{"alias_kind":"pith_short_12","alias_value":"LHUY7QBLM3RR","created_at":"2026-07-05T07:27:09.494554+00:00"},{"alias_kind":"pith_short_16","alias_value":"LHUY7QBLM3RRCXQG","created_at":"2026-07-05T07:27:09.494554+00:00"},{"alias_kind":"pith_short_8","alias_value":"LHUY7QBL","created_at":"2026-07-05T07:27:09.494554+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":7,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.20502","citing_title":"Calibration Without Comprehension: Diagnosing the Limits of Fine-Tuning LLMs for Vulnerability Detection in Systems Software","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2606.30587","citing_title":"Words Speak Louder Than Code: Investigating Cognitive Heuristics in LLM-Based Code Vulnerability Detection","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2606.29155","citing_title":"OASIF: An Efficient Obfuscation-Aware Self-Improving Framework for LLM-Based Assembly Code Instruction Following and Comprehension","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2412.11194","citing_title":"Direction for Detection: A Survey of Automated Vulnerability Detection and all of its Pain Points","ref_index":143,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18153","citing_title":"Three Heads Are Better Than One: A Multi-perspective Reasoning Framework for Enhanced Vulnerability Detection","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2510.04056","citing_title":"QuiLL: An LLM-Based Vulnerability Assessment Framework for the Wild","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2605.00072","citing_title":"XekRung Technical Report","ref_index":86,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/LHUY7QBLM3RRCXQGG2DIWUY3AE","json":"https://pith.science/pith/LHUY7QBLM3RRCXQGG2DIWUY3AE.json","graph_json":"https://pith.science/api/pith-number/LHUY7QBLM3RRCXQGG2DIWUY3AE/graph.json","events_json":"https://pith.science/api/pith-number/LHUY7QBLM3RRCXQGG2DIWUY3AE/events.json","paper":"https://pith.science/paper/LHUY7QBL"},"agent_actions":{"view_html":"https://pith.science/pith/LHUY7QBLM3RRCXQGG2DIWUY3AE","download_json":"https://pith.science/pith/LHUY7QBLM3RRCXQGG2DIWUY3AE.json","view_paper":"https://pith.science/paper/LHUY7QBL","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2311.12420&json=true","fetch_graph":"https://pith.science/api/pith-number/LHUY7QBLM3RRCXQGG2DIWUY3AE/graph.json","fetch_events":"https://pith.science/api/pith-number/LHUY7QBLM3RRCXQGG2DIWUY3AE/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/LHUY7QBLM3RRCXQGG2DIWUY3AE/action/timestamp_anchor","attest_storage":"https://pith.science/pith/LHUY7QBLM3RRCXQGG2DIWUY3AE/action/storage_attestation","attest_author":"https://pith.science/pith/LHUY7QBLM3RRCXQGG2DIWUY3AE/action/author_attestation","sign_citation":"https://pith.science/pith/LHUY7QBLM3RRCXQGG2DIWUY3AE/action/citation_signature","submit_replication":"https://pith.science/pith/LHUY7QBLM3RRCXQGG2DIWUY3AE/action/replication_record"}},"created_at":"2026-07-05T07:27:09.494554+00:00","updated_at":"2026-07-05T07:27:09.494554+00:00"}