{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:YW6LLBWCHU5W4U5WGJPVTEMN24","short_pith_number":"pith:YW6LLBWC","schema_version":"1.0","canonical_sha256":"c5bcb586c23d3b6e53b6325f59918dd721509b0588a847be3d8771abeafbc82b","source":{"kind":"arxiv","id":"2505.19828","version":1},"attestation_state":"computed","paper":{"title":"SecVulEval: Benchmarking LLMs for Real-World C/C++ Vulnerability Detection","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.SE","authors_text":"Hung Viet Pham, Jiho Shin, Md Basim Uddin Ahmed, Nima Shiri Harzevili, Song Wang","submitted_at":"2025-05-26T11:06:03Z","abstract_excerpt":"Large Language Models (LLMs) have shown promise in software engineering tasks, but evaluating their effectiveness in vulnerability detection is challenging due to the lack of high-quality datasets. Most existing datasets are limited to function-level labels, ignoring finer-grained vulnerability patterns and crucial contextual information. Also, poor data quality such as mislabeling, inconsistent annotations, and duplicates can lead to inflated performance and weak generalization. Moreover, by including only the functions, these datasets miss broader program context, like data/control dependenc"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.19828","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.SE","submitted_at":"2025-05-26T11:06:03Z","cross_cats_sorted":[],"title_canon_sha256":"d4b118f1f55e5cbf4adfacef0396eb67140f298fd81a0fdcb4d046342bae44e7","abstract_canon_sha256":"0ff5e6625c3321f8d5b743fa027e29bb92a553062f84a5ff6b6132229f7843fb"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:09:48.538293Z","signature_b64":"Dh3pi9jOOBtnNOM2SpFx6x5cknFas2xJHYsuddymwDuPQyzG7bj9YZcJHs4kFW9d/RmI+sByfZYF0y+ky5ncCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"c5bcb586c23d3b6e53b6325f59918dd721509b0588a847be3d8771abeafbc82b","last_reissued_at":"2026-07-05T11:09:48.537783Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:09:48.537783Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"SecVulEval: Benchmarking LLMs for Real-World C/C++ Vulnerability Detection","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.SE","authors_text":"Hung Viet Pham, Jiho Shin, Md Basim Uddin Ahmed, Nima Shiri Harzevili, Song Wang","submitted_at":"2025-05-26T11:06:03Z","abstract_excerpt":"Large Language Models (LLMs) have shown promise in software engineering tasks, but evaluating their effectiveness in vulnerability detection is challenging due to the lack of high-quality datasets. Most existing datasets are limited to function-level labels, ignoring finer-grained vulnerability patterns and crucial contextual information. Also, poor data quality such as mislabeling, inconsistent annotations, and duplicates can lead to inflated performance and weak generalization. Moreover, by including only the functions, these datasets miss broader program context, like data/control dependenc"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.19828","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.19828/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.19828","created_at":"2026-07-05T11:09:48.537841+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.19828v1","created_at":"2026-07-05T11:09:48.537841+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.19828","created_at":"2026-07-05T11:09:48.537841+00:00"},{"alias_kind":"pith_short_12","alias_value":"YW6LLBWCHU5W","created_at":"2026-07-05T11:09:48.537841+00:00"},{"alias_kind":"pith_short_16","alias_value":"YW6LLBWCHU5W4U5W","created_at":"2026-07-05T11:09:48.537841+00:00"},{"alias_kind":"pith_short_8","alias_value":"YW6LLBWC","created_at":"2026-07-05T11:09:48.537841+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.21397","citing_title":"Evaluating LLMs for Real-World Web Vulnerability Detection","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2606.18190","citing_title":"Multi-Source Cybersecurity Logs: An ATT&CK-Labeled Dataset and SLM Evaluation","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2606.01364","citing_title":"Needles at Scale: LLM-Assisted Target Selection for Windows Vulnerability Research","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2605.21615","citing_title":"ASSEMBLAGE-DEEPHISTORY: A Cross-Build Binary Dataset with Temporal Coverage","ref_index":2,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/YW6LLBWCHU5W4U5WGJPVTEMN24","json":"https://pith.science/pith/YW6LLBWCHU5W4U5WGJPVTEMN24.json","graph_json":"https://pith.science/api/pith-number/YW6LLBWCHU5W4U5WGJPVTEMN24/graph.json","events_json":"https://pith.science/api/pith-number/YW6LLBWCHU5W4U5WGJPVTEMN24/events.json","paper":"https://pith.science/paper/YW6LLBWC"},"agent_actions":{"view_html":"https://pith.science/pith/YW6LLBWCHU5W4U5WGJPVTEMN24","download_json":"https://pith.science/pith/YW6LLBWCHU5W4U5WGJPVTEMN24.json","view_paper":"https://pith.science/paper/YW6LLBWC","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.19828&json=true","fetch_graph":"https://pith.science/api/pith-number/YW6LLBWCHU5W4U5WGJPVTEMN24/graph.json","fetch_events":"https://pith.science/api/pith-number/YW6LLBWCHU5W4U5WGJPVTEMN24/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/YW6LLBWCHU5W4U5WGJPVTEMN24/action/timestamp_anchor","attest_storage":"https://pith.science/pith/YW6LLBWCHU5W4U5WGJPVTEMN24/action/storage_attestation","attest_author":"https://pith.science/pith/YW6LLBWCHU5W4U5WGJPVTEMN24/action/author_attestation","sign_citation":"https://pith.science/pith/YW6LLBWCHU5W4U5WGJPVTEMN24/action/citation_signature","submit_replication":"https://pith.science/pith/YW6LLBWCHU5W4U5WGJPVTEMN24/action/replication_record"}},"created_at":"2026-07-05T11:09:48.537841+00:00","updated_at":"2026-07-05T11:09:48.537841+00:00"}