{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:HU2VGYC5YAE5NONHLRTV3G2HYG","short_pith_number":"pith:HU2VGYC5","schema_version":"1.0","canonical_sha256":"3d3553605dc009d6b9a75c675d9b47c19f2c6687c1594643b8077659cfea69e3","source":{"kind":"arxiv","id":"2312.04724","version":1},"attestation_state":"computed","paper":{"title":"Purple Llama CyberSecEval: A Secure Coding Benchmark for Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CR","authors_text":"Aleksandar Straumann, Cornelius Aschermann, Cyrus Nikolaidis, Daniel Song, David LeBlanc, Dhaval Kapil, Dominik Gabi, Faizan Ahmad, Gabriel Synnaeve, Ivan Evtimov, James Milazzo, Joshua Saxe, Lorenzo Fontana, Manish Bhatt, Ravi Prakash Giri, Sahana Chennabasappa, Sasha Frolov, Shengye Wan, Spencer Whitman, Varun Vontimitta, Yiannis Kozyrakis","submitted_at":"2023-12-07T22:07:54Z","abstract_excerpt":"This paper presents CyberSecEval, a comprehensive benchmark developed to help bolster the cybersecurity of Large Language Models (LLMs) employed as coding assistants. As what we believe to be the most extensive unified cybersecurity safety benchmark to date, CyberSecEval provides a thorough evaluation of LLMs in two crucial security domains: their propensity to generate insecure code and their level of compliance when asked to assist in cyberattacks. Through a case study involving seven models from the Llama 2, Code Llama, and OpenAI GPT large language model families, CyberSecEval effectively "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2312.04724","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CR","submitted_at":"2023-12-07T22:07:54Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"9dd782649300bbc3bbfed4425a7a1ccbe4429f76f6f40716d4e48c35ee66b3d0","abstract_canon_sha256":"c39c41c73f7fae133226c05c6994d1c11262d31b9cea67e0cdda63fece47def1"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:21:50.425050Z","signature_b64":"e9k0p6RhXV94hymaR+kt2MA/msebDk1W6XXfVAGmKdGDF9CZwU5BzY+rMavWGuXRchUONsmXkGez97Hgt7CECQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"3d3553605dc009d6b9a75c675d9b47c19f2c6687c1594643b8077659cfea69e3","last_reissued_at":"2026-07-05T07:21:50.424526Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:21:50.424526Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Purple Llama CyberSecEval: A Secure Coding Benchmark for Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CR","authors_text":"Aleksandar Straumann, Cornelius Aschermann, Cyrus Nikolaidis, Daniel Song, David LeBlanc, Dhaval Kapil, Dominik Gabi, Faizan Ahmad, Gabriel Synnaeve, Ivan Evtimov, James Milazzo, Joshua Saxe, Lorenzo Fontana, Manish Bhatt, Ravi Prakash Giri, Sahana Chennabasappa, Sasha Frolov, Shengye Wan, Spencer Whitman, Varun Vontimitta, Yiannis Kozyrakis","submitted_at":"2023-12-07T22:07:54Z","abstract_excerpt":"This paper presents CyberSecEval, a comprehensive benchmark developed to help bolster the cybersecurity of Large Language Models (LLMs) employed as coding assistants. As what we believe to be the most extensive unified cybersecurity safety benchmark to date, CyberSecEval provides a thorough evaluation of LLMs in two crucial security domains: their propensity to generate insecure code and their level of compliance when asked to assist in cyberattacks. Through a case study involving seven models from the Llama 2, Code Llama, and OpenAI GPT large language model families, CyberSecEval effectively "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2312.04724","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2312.04724/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2312.04724","created_at":"2026-07-05T07:21:50.424583+00:00"},{"alias_kind":"arxiv_version","alias_value":"2312.04724v1","created_at":"2026-07-05T07:21:50.424583+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2312.04724","created_at":"2026-07-05T07:21:50.424583+00:00"},{"alias_kind":"pith_short_12","alias_value":"HU2VGYC5YAE5","created_at":"2026-07-05T07:21:50.424583+00:00"},{"alias_kind":"pith_short_16","alias_value":"HU2VGYC5YAE5NONH","created_at":"2026-07-05T07:21:50.424583+00:00"},{"alias_kind":"pith_short_8","alias_value":"HU2VGYC5","created_at":"2026-07-05T07:21:50.424583+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":23,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.25973","citing_title":"Helpful or Harmful? Evaluating LLM-Assisted Vulnerability Patching via a Human Study","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2606.01317","citing_title":"SABER: Benchmarking Operational Safety of LLM Coding Agents in Stateful Project Workspaces","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2606.29175","citing_title":"Direct Causation in International Humanitarian Law and the Challenge of AI-Mediated Civilian Cyber Operations","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2605.22643","citing_title":"Boiling the Frog: A Multi-Turn Benchmark for Agentic Safety","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2605.23091","citing_title":"Security of LLM-generated Code: A Comparative Analysis","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2411.10656","citing_title":"Precision or Peril: A PoC of Python Code Quality from Quantized Large Language Models","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2605.21773","citing_title":"HIDBench: Benchmarking Large Language Models for Host-Based Intrusion Detection","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2605.22643","citing_title":"Boiling the Frog: A Multi-Turn Benchmark for Agentic Safety","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2605.20351","citing_title":"Refusal Evaluation in Coding LLMs and Code Agents: A Systematic Review of Thirteen Malicious-Code Prompt Corpora (2023-2025)","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17413","citing_title":"Ablating Safety: Mechanisms for Removing Alignment in Language Models for Security Applications","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2512.05439","citing_title":"BEAVER: An Efficient Deterministic LLM Verifier","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2602.06759","citing_title":"\"Tab, Tab, Bug\": Security Pitfalls of Next Edit Suggestions in AI-Integrated IDEs","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2402.19173","citing_title":"StarCoder 2 and The Stack v2: The Next Generation","ref_index":164,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08382","citing_title":"SecureForge: Finding and Preventing Vulnerabilities in LLM-Generated Code via Prompt Optimization","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08898","citing_title":"LLM-Agnostic Semantic Representation Attack","ref_index":76,"is_internal_anchor":false},{"citing_arxiv_id":"2605.04019","citing_title":"Redefining AI Red Teaming in the Agentic Era: From Weeks to Hours","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05267","citing_title":"Bridging Generation and Training: A Systematic Review of Quality Issues in LLMs for Code","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2604.18718","citing_title":"Towards Optimal Agentic Architectures for Offensive Security Tasks","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2604.09544","citing_title":"Large Language Models Generate Harmful Responses Using a Distinct Mechanism, Shared Across Harm Types","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2604.05292","citing_title":"Broken by Default: A Formal Verification Study of Security Vulnerabilities in AI-Generated Code","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2501.14249","citing_title":"Humanity's Last Exam","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2604.17803","citing_title":"Adversarial Arena: Crowdsourcing Data Generation through Interactive Competition","ref_index":52,"is_internal_anchor":false},{"citing_arxiv_id":"2605.03179","citing_title":"A Validated Prompt Bank for Malicious Code Generation: Separating Executable Weapons from Security Knowledge in 1,554 Consensus-Labeled Prompts","ref_index":47,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/HU2VGYC5YAE5NONHLRTV3G2HYG","json":"https://pith.science/pith/HU2VGYC5YAE5NONHLRTV3G2HYG.json","graph_json":"https://pith.science/api/pith-number/HU2VGYC5YAE5NONHLRTV3G2HYG/graph.json","events_json":"https://pith.science/api/pith-number/HU2VGYC5YAE5NONHLRTV3G2HYG/events.json","paper":"https://pith.science/paper/HU2VGYC5"},"agent_actions":{"view_html":"https://pith.science/pith/HU2VGYC5YAE5NONHLRTV3G2HYG","download_json":"https://pith.science/pith/HU2VGYC5YAE5NONHLRTV3G2HYG.json","view_paper":"https://pith.science/paper/HU2VGYC5","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2312.04724&json=true","fetch_graph":"https://pith.science/api/pith-number/HU2VGYC5YAE5NONHLRTV3G2HYG/graph.json","fetch_events":"https://pith.science/api/pith-number/HU2VGYC5YAE5NONHLRTV3G2HYG/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/HU2VGYC5YAE5NONHLRTV3G2HYG/action/timestamp_anchor","attest_storage":"https://pith.science/pith/HU2VGYC5YAE5NONHLRTV3G2HYG/action/storage_attestation","attest_author":"https://pith.science/pith/HU2VGYC5YAE5NONHLRTV3G2HYG/action/author_attestation","sign_citation":"https://pith.science/pith/HU2VGYC5YAE5NONHLRTV3G2HYG/action/citation_signature","submit_replication":"https://pith.science/pith/HU2VGYC5YAE5NONHLRTV3G2HYG/action/replication_record"}},"created_at":"2026-07-05T07:21:50.424583+00:00","updated_at":"2026-07-05T07:21:50.424583+00:00"}