{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:7CLUVVHMVPGGC4CDJEDQLTEKN3","short_pith_number":"pith:7CLUVVHM","schema_version":"1.0","canonical_sha256":"f8974ad4ecabcc617043490705cc8a6ecda8e9363dd70b21cda6fd711a8e8d15","source":{"kind":"arxiv","id":"2312.15838","version":1},"attestation_state":"computed","paper":{"title":"SecQA: A Concise Question-Answering Dataset for Evaluating Large Language Models in Computer Security","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CR"],"primary_cat":"cs.CL","authors_text":"Zefang Liu","submitted_at":"2023-12-26T00:59:30Z","abstract_excerpt":"In this paper, we introduce SecQA, a novel dataset tailored for evaluating the performance of Large Language Models (LLMs) in the domain of computer security. Utilizing multiple-choice questions generated by GPT-4 based on the \"Computer Systems Security: Planning for Success\" textbook, SecQA aims to assess LLMs' understanding and application of security principles. We detail the structure and intent of SecQA, which includes two versions of increasing complexity, to provide a concise evaluation across various difficulty levels. Additionally, we present an extensive evaluation of prominent LLMs,"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2312.15838","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2023-12-26T00:59:30Z","cross_cats_sorted":["cs.CR"],"title_canon_sha256":"e6e8ad05b8bd6c400ebe1cbb2cbf34ec3d1df72c6bf4ce9328bef42c953c41a7","abstract_canon_sha256":"d698bf2a88b5a0f7f995102ce3555b99796401c6d32c2a63395f8f2b63022258"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:28:02.137040Z","signature_b64":"0CCiMlxmIqKR5oJl+/PT1bsrPT32eYxZ5cUtZQQWxfbcohsdJYHhNxsPKl8N+Fty1sdEA6nAQCdX4l9mURQcBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f8974ad4ecabcc617043490705cc8a6ecda8e9363dd70b21cda6fd711a8e8d15","last_reissued_at":"2026-07-05T07:28:02.136700Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:28:02.136700Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"SecQA: A Concise Question-Answering Dataset for Evaluating Large Language Models in Computer Security","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CR"],"primary_cat":"cs.CL","authors_text":"Zefang Liu","submitted_at":"2023-12-26T00:59:30Z","abstract_excerpt":"In this paper, we introduce SecQA, a novel dataset tailored for evaluating the performance of Large Language Models (LLMs) in the domain of computer security. Utilizing multiple-choice questions generated by GPT-4 based on the \"Computer Systems Security: Planning for Success\" textbook, SecQA aims to assess LLMs' understanding and application of security principles. We detail the structure and intent of SecQA, which includes two versions of increasing complexity, to provide a concise evaluation across various difficulty levels. Additionally, we present an extensive evaluation of prominent LLMs,"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2312.15838","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2312.15838/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2312.15838","created_at":"2026-07-05T07:28:02.136748+00:00"},{"alias_kind":"arxiv_version","alias_value":"2312.15838v1","created_at":"2026-07-05T07:28:02.136748+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2312.15838","created_at":"2026-07-05T07:28:02.136748+00:00"},{"alias_kind":"pith_short_12","alias_value":"7CLUVVHMVPGG","created_at":"2026-07-05T07:28:02.136748+00:00"},{"alias_kind":"pith_short_16","alias_value":"7CLUVVHMVPGGC4CD","created_at":"2026-07-05T07:28:02.136748+00:00"},{"alias_kind":"pith_short_8","alias_value":"7CLUVVHM","created_at":"2026-07-05T07:28:02.136748+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.24765","citing_title":"CyberMaskQA: A Privacy-Aware Benchmark for Evaluating Large Language Models in Cybersecurity Question Answering","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2605.28146","citing_title":"Cybersecurity AI (CAI) Dataset","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2507.14201","citing_title":"ExCyTIn-Bench: Evaluating LLM agents on Cyber Threat Investigation","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2602.05523","citing_title":"Capture the Flags: Family-Based Evaluation of Agentic LLMs via Semantics-Preserving Transformations","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2604.12040","citing_title":"SIR-Bench: Evaluating Investigation Depth in Security Incident Response Agents","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2604.20389","citing_title":"CyberCertBench: Evaluating LLMs in Cybersecurity Certification Knowledge","ref_index":16,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/7CLUVVHMVPGGC4CDJEDQLTEKN3","json":"https://pith.science/pith/7CLUVVHMVPGGC4CDJEDQLTEKN3.json","graph_json":"https://pith.science/api/pith-number/7CLUVVHMVPGGC4CDJEDQLTEKN3/graph.json","events_json":"https://pith.science/api/pith-number/7CLUVVHMVPGGC4CDJEDQLTEKN3/events.json","paper":"https://pith.science/paper/7CLUVVHM"},"agent_actions":{"view_html":"https://pith.science/pith/7CLUVVHMVPGGC4CDJEDQLTEKN3","download_json":"https://pith.science/pith/7CLUVVHMVPGGC4CDJEDQLTEKN3.json","view_paper":"https://pith.science/paper/7CLUVVHM","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2312.15838&json=true","fetch_graph":"https://pith.science/api/pith-number/7CLUVVHMVPGGC4CDJEDQLTEKN3/graph.json","fetch_events":"https://pith.science/api/pith-number/7CLUVVHMVPGGC4CDJEDQLTEKN3/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/7CLUVVHMVPGGC4CDJEDQLTEKN3/action/timestamp_anchor","attest_storage":"https://pith.science/pith/7CLUVVHMVPGGC4CDJEDQLTEKN3/action/storage_attestation","attest_author":"https://pith.science/pith/7CLUVVHMVPGGC4CDJEDQLTEKN3/action/author_attestation","sign_citation":"https://pith.science/pith/7CLUVVHMVPGGC4CDJEDQLTEKN3/action/citation_signature","submit_replication":"https://pith.science/pith/7CLUVVHMVPGGC4CDJEDQLTEKN3/action/replication_record"}},"created_at":"2026-07-05T07:28:02.136748+00:00","updated_at":"2026-07-05T07:28:02.136748+00:00"}