{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:4RCPB72FYDO6TN4GQFGSLSYQZN","short_pith_number":"pith:4RCPB72F","schema_version":"1.0","canonical_sha256":"e444f0ff45c0dde9b786814d25cb10cb7c74081279481ed9ebf61109d7f09b6e","source":{"kind":"arxiv","id":"2408.13204","version":1},"attestation_state":"computed","paper":{"title":"DOMAINEVAL: An Auto-Constructed Benchmark for Multi-Domain Code Generation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.SE"],"primary_cat":"cs.AI","authors_text":"Hongyu Lin, Jialun Cao, Le Sun, Qiming Zhu, Shing-Chi Cheung, Xianpei Han, Yaojie Lu","submitted_at":"2024-08-23T16:33:58Z","abstract_excerpt":"Code benchmarks such as HumanEval are widely adopted to evaluate the capabilities of Large Language Models (LLMs), providing insights into their strengths and weaknesses. However, current benchmarks primarily exercise LLMs' capability on common coding tasks (e.g., bubble sort, greatest common divisor), leaving domain-specific coding tasks (e.g., computation, system, cryptography) unexplored. To fill this gap, we propose a multi-domain code benchmark, DOMAINEVAL, designed to evaluate LLMs' coding capabilities thoroughly. Our pipeline works in a fully automated manner, enabling a push-bottom con"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2408.13204","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2024-08-23T16:33:58Z","cross_cats_sorted":["cs.SE"],"title_canon_sha256":"5551fd068d2c5cf38e0eb0422ac891a27bc44c05ce300634195c5a7c09d5a0d9","abstract_canon_sha256":"6e04cb793a05182954ca3c3c28f6852e2750f610a0c0769c0e4ad932ff5e4f8a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:58:36.897189Z","signature_b64":"Kq/JF77+rSAPh4MorYIRub95v9vrfOhyUN2eXVQGdSP4nWlqTAHSFV255z9hzeFAUDZrJ0QGBqrrnF3k/7CXBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"e444f0ff45c0dde9b786814d25cb10cb7c74081279481ed9ebf61109d7f09b6e","last_reissued_at":"2026-07-05T08:58:36.896774Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:58:36.896774Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"DOMAINEVAL: An Auto-Constructed Benchmark for Multi-Domain Code Generation","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.SE"],"primary_cat":"cs.AI","authors_text":"Hongyu Lin, Jialun Cao, Le Sun, Qiming Zhu, Shing-Chi Cheung, Xianpei Han, Yaojie Lu","submitted_at":"2024-08-23T16:33:58Z","abstract_excerpt":"Code benchmarks such as HumanEval are widely adopted to evaluate the capabilities of Large Language Models (LLMs), providing insights into their strengths and weaknesses. However, current benchmarks primarily exercise LLMs' capability on common coding tasks (e.g., bubble sort, greatest common divisor), leaving domain-specific coding tasks (e.g., computation, system, cryptography) unexplored. To fill this gap, we propose a multi-domain code benchmark, DOMAINEVAL, designed to evaluate LLMs' coding capabilities thoroughly. Our pipeline works in a fully automated manner, enabling a push-bottom con"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2408.13204","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2408.13204/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2408.13204","created_at":"2026-07-05T08:58:36.896833+00:00"},{"alias_kind":"arxiv_version","alias_value":"2408.13204v1","created_at":"2026-07-05T08:58:36.896833+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2408.13204","created_at":"2026-07-05T08:58:36.896833+00:00"},{"alias_kind":"pith_short_12","alias_value":"4RCPB72FYDO6","created_at":"2026-07-05T08:58:36.896833+00:00"},{"alias_kind":"pith_short_16","alias_value":"4RCPB72FYDO6TN4G","created_at":"2026-07-05T08:58:36.896833+00:00"},{"alias_kind":"pith_short_8","alias_value":"4RCPB72F","created_at":"2026-07-05T08:58:36.896833+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2604.26923","citing_title":"ClassEval-Pro: A Cross-Domain Benchmark for Class-Level Code Generation","ref_index":52,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/4RCPB72FYDO6TN4GQFGSLSYQZN","json":"https://pith.science/pith/4RCPB72FYDO6TN4GQFGSLSYQZN.json","graph_json":"https://pith.science/api/pith-number/4RCPB72FYDO6TN4GQFGSLSYQZN/graph.json","events_json":"https://pith.science/api/pith-number/4RCPB72FYDO6TN4GQFGSLSYQZN/events.json","paper":"https://pith.science/paper/4RCPB72F"},"agent_actions":{"view_html":"https://pith.science/pith/4RCPB72FYDO6TN4GQFGSLSYQZN","download_json":"https://pith.science/pith/4RCPB72FYDO6TN4GQFGSLSYQZN.json","view_paper":"https://pith.science/paper/4RCPB72F","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2408.13204&json=true","fetch_graph":"https://pith.science/api/pith-number/4RCPB72FYDO6TN4GQFGSLSYQZN/graph.json","fetch_events":"https://pith.science/api/pith-number/4RCPB72FYDO6TN4GQFGSLSYQZN/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/4RCPB72FYDO6TN4GQFGSLSYQZN/action/timestamp_anchor","attest_storage":"https://pith.science/pith/4RCPB72FYDO6TN4GQFGSLSYQZN/action/storage_attestation","attest_author":"https://pith.science/pith/4RCPB72FYDO6TN4GQFGSLSYQZN/action/author_attestation","sign_citation":"https://pith.science/pith/4RCPB72FYDO6TN4GQFGSLSYQZN/action/citation_signature","submit_replication":"https://pith.science/pith/4RCPB72FYDO6TN4GQFGSLSYQZN/action/replication_record"}},"created_at":"2026-07-05T08:58:36.896833+00:00","updated_at":"2026-07-05T08:58:36.896833+00:00"}