{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:4Z7RCIJO5XK3O44B33B3EKS2SZ","short_pith_number":"pith:4Z7RCIJO","schema_version":"1.0","canonical_sha256":"e67f11212eedd5b77381dec3b22a5a9656c4eba7fd97fbd5376a4d8fb04b48c5","source":{"kind":"arxiv","id":"2501.08200","version":1},"attestation_state":"computed","paper":{"title":"CWEval: Outcome-driven Evaluation on Functionality and Security of LLM Code Generation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL","cs.LG"],"primary_cat":"cs.SE","authors_text":"Baishakhi Ray, Jinjun Peng, Junfeng Yang, Kele Huang, Leyi Cui","submitted_at":"2025-01-14T15:27:01Z","abstract_excerpt":"Large Language Models (LLMs) have significantly aided developers by generating or assisting in code writing, enhancing productivity across various tasks. While identifying incorrect code is often straightforward, detecting vulnerabilities in functionally correct code is more challenging, especially for developers with limited security knowledge, which poses considerable security risks of using LLM-generated code and underscores the need for robust evaluation benchmarks that assess both functional correctness and security. Current benchmarks like CyberSecEval and SecurityEval attempt to solve i"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2501.08200","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.SE","submitted_at":"2025-01-14T15:27:01Z","cross_cats_sorted":["cs.CL","cs.LG"],"title_canon_sha256":"b00360a3bc4b1d220d19debd964ac0964b5fbedb8fa6ce9ccb4e2a8a1695f362","abstract_canon_sha256":"2b31cd1dcfe0149a0e362581455c87dd17e9dffe302ed67161e976a41814f1f3"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:00:57.453867Z","signature_b64":"iowe0nztGLUN8JpjDaFKtyUJxc2hC2uEQwyPWM/smegOVMIfGpeaSbhgkym8fw3V522YtCtIROkg3CXJVoYFAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"e67f11212eedd5b77381dec3b22a5a9656c4eba7fd97fbd5376a4d8fb04b48c5","last_reissued_at":"2026-07-05T10:00:57.453291Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:00:57.453291Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"CWEval: Outcome-driven Evaluation on Functionality and Security of LLM Code Generation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL","cs.LG"],"primary_cat":"cs.SE","authors_text":"Baishakhi Ray, Jinjun Peng, Junfeng Yang, Kele Huang, Leyi Cui","submitted_at":"2025-01-14T15:27:01Z","abstract_excerpt":"Large Language Models (LLMs) have significantly aided developers by generating or assisting in code writing, enhancing productivity across various tasks. While identifying incorrect code is often straightforward, detecting vulnerabilities in functionally correct code is more challenging, especially for developers with limited security knowledge, which poses considerable security risks of using LLM-generated code and underscores the need for robust evaluation benchmarks that assess both functional correctness and security. Current benchmarks like CyberSecEval and SecurityEval attempt to solve i"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2501.08200","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2501.08200/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2501.08200","created_at":"2026-07-05T10:00:57.453375+00:00"},{"alias_kind":"arxiv_version","alias_value":"2501.08200v1","created_at":"2026-07-05T10:00:57.453375+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2501.08200","created_at":"2026-07-05T10:00:57.453375+00:00"},{"alias_kind":"pith_short_12","alias_value":"4Z7RCIJO5XK3","created_at":"2026-07-05T10:00:57.453375+00:00"},{"alias_kind":"pith_short_16","alias_value":"4Z7RCIJO5XK3O44B","created_at":"2026-07-05T10:00:57.453375+00:00"},{"alias_kind":"pith_short_8","alias_value":"4Z7RCIJO","created_at":"2026-07-05T10:00:57.453375+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.25195","citing_title":"SoK: AI Secure Code Generation: Progress, Pitfalls, and Paths Forward","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2606.04460","citing_title":"CyberGym-E2E: Scalable Real-World Benchmark for AI Agents' End-to-End Cybersecurity Capabilities","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2503.14281","citing_title":"XOXO: Stealthy Cross-Origin Context Poisoning Attacks against AI Coding Assistants","ref_index":49,"is_internal_anchor":false},{"citing_arxiv_id":"2512.21132","citing_title":"AutoBaxBuilder: Bootstrapping Code Security Benchmarking","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2605.16829","citing_title":"Constrained Code Generation with Discrete Diffusion","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2604.17014","citing_title":"False Security Confidence in Benign LLM Code Generation","ref_index":4,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/4Z7RCIJO5XK3O44B33B3EKS2SZ","json":"https://pith.science/pith/4Z7RCIJO5XK3O44B33B3EKS2SZ.json","graph_json":"https://pith.science/api/pith-number/4Z7RCIJO5XK3O44B33B3EKS2SZ/graph.json","events_json":"https://pith.science/api/pith-number/4Z7RCIJO5XK3O44B33B3EKS2SZ/events.json","paper":"https://pith.science/paper/4Z7RCIJO"},"agent_actions":{"view_html":"https://pith.science/pith/4Z7RCIJO5XK3O44B33B3EKS2SZ","download_json":"https://pith.science/pith/4Z7RCIJO5XK3O44B33B3EKS2SZ.json","view_paper":"https://pith.science/paper/4Z7RCIJO","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2501.08200&json=true","fetch_graph":"https://pith.science/api/pith-number/4Z7RCIJO5XK3O44B33B3EKS2SZ/graph.json","fetch_events":"https://pith.science/api/pith-number/4Z7RCIJO5XK3O44B33B3EKS2SZ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/4Z7RCIJO5XK3O44B33B3EKS2SZ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/4Z7RCIJO5XK3O44B33B3EKS2SZ/action/storage_attestation","attest_author":"https://pith.science/pith/4Z7RCIJO5XK3O44B33B3EKS2SZ/action/author_attestation","sign_citation":"https://pith.science/pith/4Z7RCIJO5XK3O44B33B3EKS2SZ/action/citation_signature","submit_replication":"https://pith.science/pith/4Z7RCIJO5XK3O44B33B3EKS2SZ/action/replication_record"}},"created_at":"2026-07-05T10:00:57.453375+00:00","updated_at":"2026-07-05T10:00:57.453375+00:00"}