{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:X7K4X7FF4HF465RW7U6HJIXDUB","short_pith_number":"pith:X7K4X7FF","schema_version":"1.0","canonical_sha256":"bfd5cbfca5e1cbcf7636fd3c74a2e3a0490527cbaaf89cafd9e6cfe1d54e4e0d","source":{"kind":"arxiv","id":"2404.00566","version":4},"attestation_state":"computed","paper":{"title":"CodeBenchGen: Creating Scalable Execution-based Code Generation Benchmarks","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.SE","authors_text":"Alex Xie, Carolyn Rose, Daniel Fried, Divyanshu Sheth, Pengfei Liu, Yiqing Xie","submitted_at":"2024-03-31T05:20:53Z","abstract_excerpt":"To adequately test modern code generation systems, evaluation benchmarks must execute and test the code generated by the system. However, these execution and testing requirements have largely limited benchmarks to settings where code is easily executable or has human-written tests. To facilitate evaluation of code generation systems across diverse scenarios, we present CodeBenchGen, a framework to create scalable execution-based benchmarks from naturally occurring code sources. Specifically, we leverage a large language model (LLM) to sandbox arbitrary pieces of code into evaluation examples, "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2404.00566","kind":"arxiv","version":4},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.SE","submitted_at":"2024-03-31T05:20:53Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"e83fb6794ef3a99470b6e51fc476d9052290c9036d0fd8c60947ae2a189fdc51","abstract_canon_sha256":"5af516f9cdd9e26dee3560a8760fce55c303a6f09fe834c0bd948b494d7f5bbc"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:14:54.113611Z","signature_b64":"6PvmXzn0Ck7W37n56Y3H8DTdVP5A/2dsWWtMc7bcQ14Ij+J4sP8dpcywKKxNnvij8pbR93LBz3e/vj3rTIDjAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"bfd5cbfca5e1cbcf7636fd3c74a2e3a0490527cbaaf89cafd9e6cfe1d54e4e0d","last_reissued_at":"2026-07-05T09:14:54.113169Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:14:54.113169Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"CodeBenchGen: Creating Scalable Execution-based Code Generation Benchmarks","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.SE","authors_text":"Alex Xie, Carolyn Rose, Daniel Fried, Divyanshu Sheth, Pengfei Liu, Yiqing Xie","submitted_at":"2024-03-31T05:20:53Z","abstract_excerpt":"To adequately test modern code generation systems, evaluation benchmarks must execute and test the code generated by the system. However, these execution and testing requirements have largely limited benchmarks to settings where code is easily executable or has human-written tests. To facilitate evaluation of code generation systems across diverse scenarios, we present CodeBenchGen, a framework to create scalable execution-based benchmarks from naturally occurring code sources. Specifically, we leverage a large language model (LLM) to sandbox arbitrary pieces of code into evaluation examples, "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2404.00566","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2404.00566/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2404.00566","created_at":"2026-07-05T09:14:54.113232+00:00"},{"alias_kind":"arxiv_version","alias_value":"2404.00566v4","created_at":"2026-07-05T09:14:54.113232+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2404.00566","created_at":"2026-07-05T09:14:54.113232+00:00"},{"alias_kind":"pith_short_12","alias_value":"X7K4X7FF4HF4","created_at":"2026-07-05T09:14:54.113232+00:00"},{"alias_kind":"pith_short_16","alias_value":"X7K4X7FF4HF465RW","created_at":"2026-07-05T09:14:54.113232+00:00"},{"alias_kind":"pith_short_8","alias_value":"X7K4X7FF","created_at":"2026-07-05T09:14:54.113232+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/X7K4X7FF4HF465RW7U6HJIXDUB","json":"https://pith.science/pith/X7K4X7FF4HF465RW7U6HJIXDUB.json","graph_json":"https://pith.science/api/pith-number/X7K4X7FF4HF465RW7U6HJIXDUB/graph.json","events_json":"https://pith.science/api/pith-number/X7K4X7FF4HF465RW7U6HJIXDUB/events.json","paper":"https://pith.science/paper/X7K4X7FF"},"agent_actions":{"view_html":"https://pith.science/pith/X7K4X7FF4HF465RW7U6HJIXDUB","download_json":"https://pith.science/pith/X7K4X7FF4HF465RW7U6HJIXDUB.json","view_paper":"https://pith.science/paper/X7K4X7FF","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2404.00566&json=true","fetch_graph":"https://pith.science/api/pith-number/X7K4X7FF4HF465RW7U6HJIXDUB/graph.json","fetch_events":"https://pith.science/api/pith-number/X7K4X7FF4HF465RW7U6HJIXDUB/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/X7K4X7FF4HF465RW7U6HJIXDUB/action/timestamp_anchor","attest_storage":"https://pith.science/pith/X7K4X7FF4HF465RW7U6HJIXDUB/action/storage_attestation","attest_author":"https://pith.science/pith/X7K4X7FF4HF465RW7U6HJIXDUB/action/author_attestation","sign_citation":"https://pith.science/pith/X7K4X7FF4HF465RW7U6HJIXDUB/action/citation_signature","submit_replication":"https://pith.science/pith/X7K4X7FF4HF465RW7U6HJIXDUB/action/replication_record"}},"created_at":"2026-07-05T09:14:54.113232+00:00","updated_at":"2026-07-05T09:14:54.113232+00:00"}