{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:FTNTHYJC5NFUE6BHSILDNM6BHS","short_pith_number":"pith:FTNTHYJC","schema_version":"1.0","canonical_sha256":"2cdb33e122eb4b427827921636b3c13c98df2295c8042475938dff66a1facf2d","source":{"kind":"arxiv","id":"2402.16694","version":2},"attestation_state":"computed","paper":{"title":"HumanEval-XL: A Multilingual Code Generation Benchmark for Cross-lingual Natural Language Generalization","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.PL","cs.SE"],"primary_cat":"cs.CL","authors_text":"Qiwei Peng, Xuhong Li, Yekun Chai","submitted_at":"2024-02-26T16:09:00Z","abstract_excerpt":"Large language models (LLMs) have made significant progress in generating codes from textual prompts. However, existing benchmarks have mainly concentrated on translating English prompts to multilingual codes or have been constrained to very limited natural languages (NLs). These benchmarks have overlooked the vast landscape of massively multilingual NL to multilingual code, leaving a critical gap in the evaluation of multilingual LLMs. In response, we introduce HumanEval-XL, a massively multilingual code generation benchmark specifically crafted to address this deficiency. HumanEval-XL establ"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2402.16694","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-02-26T16:09:00Z","cross_cats_sorted":["cs.PL","cs.SE"],"title_canon_sha256":"e596d4dec99582ba8ce93c471f0d85776a2cfe3ee954a4ff2d6e19ad829e4aec","abstract_canon_sha256":"08cf1e9f7993eea283b048d17d149990e3a8571a43e35daa6593e7a32fdef5be"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:00:01.962861Z","signature_b64":"RJTbMLSUcrMpm32DIOrHSiUecJ7fni7gh0aQ2Fdqn8WAs7+r0GjKiKlhJ9DLp2sogOlsTZ3eUXou/mZRZnPZCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"2cdb33e122eb4b427827921636b3c13c98df2295c8042475938dff66a1facf2d","last_reissued_at":"2026-07-05T08:00:01.962364Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:00:01.962364Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"HumanEval-XL: A Multilingual Code Generation Benchmark for Cross-lingual Natural Language Generalization","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.PL","cs.SE"],"primary_cat":"cs.CL","authors_text":"Qiwei Peng, Xuhong Li, Yekun Chai","submitted_at":"2024-02-26T16:09:00Z","abstract_excerpt":"Large language models (LLMs) have made significant progress in generating codes from textual prompts. However, existing benchmarks have mainly concentrated on translating English prompts to multilingual codes or have been constrained to very limited natural languages (NLs). These benchmarks have overlooked the vast landscape of massively multilingual NL to multilingual code, leaving a critical gap in the evaluation of multilingual LLMs. In response, we introduce HumanEval-XL, a massively multilingual code generation benchmark specifically crafted to address this deficiency. HumanEval-XL establ"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2402.16694","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2402.16694/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2402.16694","created_at":"2026-07-05T08:00:01.962428+00:00"},{"alias_kind":"arxiv_version","alias_value":"2402.16694v2","created_at":"2026-07-05T08:00:01.962428+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2402.16694","created_at":"2026-07-05T08:00:01.962428+00:00"},{"alias_kind":"pith_short_12","alias_value":"FTNTHYJC5NFU","created_at":"2026-07-05T08:00:01.962428+00:00"},{"alias_kind":"pith_short_16","alias_value":"FTNTHYJC5NFUE6BH","created_at":"2026-07-05T08:00:01.962428+00:00"},{"alias_kind":"pith_short_8","alias_value":"FTNTHYJC","created_at":"2026-07-05T08:00:01.962428+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":8,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.06411","citing_title":"RuBench: A Repository-Level Agentic Coding Benchmark with Natively Authored Russian Task Specifications","ref_index":29,"is_internal_anchor":true},{"citing_arxiv_id":"2606.20517","citing_title":"Multi-LCB: Extending LiveCodeBench to Multiple Programming Languages","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2512.14018","citing_title":"PerfCoder: Large Language Models for Interpretable Code Performance Optimization","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2604.16321","citing_title":"LLM-Based Multi-Agent Systems for Code Generation: A Multi-Vocal Literature Review","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2604.25960","citing_title":"Large Language Models for Multilingual Code Intelligence: A Survey","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2604.14210","citing_title":"Chinese Language Is Not More Efficient Than English in Vibe Coding: A Preliminary Study on Token Cost and Problem-Solving Rate","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2604.06253","citing_title":"FLeX: Fourier-based Low-rank EXpansion for multilingual transfer","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2604.16625","citing_title":"AdaExplore: Failure-Driven Adaptation and Diversity-Preserving Search for Efficient Kernel Generation","ref_index":15,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/FTNTHYJC5NFUE6BHSILDNM6BHS","json":"https://pith.science/pith/FTNTHYJC5NFUE6BHSILDNM6BHS.json","graph_json":"https://pith.science/api/pith-number/FTNTHYJC5NFUE6BHSILDNM6BHS/graph.json","events_json":"https://pith.science/api/pith-number/FTNTHYJC5NFUE6BHSILDNM6BHS/events.json","paper":"https://pith.science/paper/FTNTHYJC"},"agent_actions":{"view_html":"https://pith.science/pith/FTNTHYJC5NFUE6BHSILDNM6BHS","download_json":"https://pith.science/pith/FTNTHYJC5NFUE6BHSILDNM6BHS.json","view_paper":"https://pith.science/paper/FTNTHYJC","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2402.16694&json=true","fetch_graph":"https://pith.science/api/pith-number/FTNTHYJC5NFUE6BHSILDNM6BHS/graph.json","fetch_events":"https://pith.science/api/pith-number/FTNTHYJC5NFUE6BHSILDNM6BHS/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/FTNTHYJC5NFUE6BHSILDNM6BHS/action/timestamp_anchor","attest_storage":"https://pith.science/pith/FTNTHYJC5NFUE6BHSILDNM6BHS/action/storage_attestation","attest_author":"https://pith.science/pith/FTNTHYJC5NFUE6BHSILDNM6BHS/action/author_attestation","sign_citation":"https://pith.science/pith/FTNTHYJC5NFUE6BHSILDNM6BHS/action/citation_signature","submit_replication":"https://pith.science/pith/FTNTHYJC5NFUE6BHSILDNM6BHS/action/replication_record"}},"created_at":"2026-07-05T08:00:01.962428+00:00","updated_at":"2026-07-05T08:00:01.962428+00:00"}