{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:W6PC23N772VYILUEPQULH4MNZG","short_pith_number":"pith:W6PC23N7","schema_version":"1.0","canonical_sha256":"b79e2d6dbffeab842e847c28b3f18dc9ae43e1e881bf1dc9059f71e3681a3f11","source":{"kind":"arxiv","id":"2303.03004","version":4},"attestation_state":"computed","paper":{"title":"xCodeEval: A Large Scale Multilingual Multitask Benchmark for Code Understanding, Generation, Translation and Retrieval","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Md Rizwan Parvez, Mohammad Abdullah Matin Khan, M Saiful Bari, Shafiq Joty, Weishi Wang, Xuan Long Do","submitted_at":"2023-03-06T10:08:51Z","abstract_excerpt":"Recently, pre-trained large language models (LLMs) have shown impressive abilities in generating codes from natural language descriptions, repairing buggy codes, translating codes between languages, and retrieving relevant code segments. However, the evaluation of these models has often been performed in a scattered way on only one or two specific tasks, in a few languages, at a partial granularity (e.g., function) level, and in many cases without proper training data. Even more concerning is that in most cases the evaluation of generated codes has been done in terms of mere lexical overlap wi"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2303.03004","kind":"arxiv","version":4},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2023-03-06T10:08:51Z","cross_cats_sorted":[],"title_canon_sha256":"21dba1c7e9763bfc6d90b408aaf55c0fae93c867855d5d690574ca7b85ac0021","abstract_canon_sha256":"5fdb78dfb718257bf6c7fffc2e2a1648437cece4447f6e185de4955d54c359b9"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:08:55.371472Z","signature_b64":"/yh8I7DqoXIRy7GqOQ7h4ZDOW8drQs1bDTglb9asOOkdI8FJeig+P06IrZOKVKGnYFTAU2qZfWm7PqOMZSzADw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b79e2d6dbffeab842e847c28b3f18dc9ae43e1e881bf1dc9059f71e3681a3f11","last_reissued_at":"2026-07-05T07:08:55.371006Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:08:55.371006Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"xCodeEval: A Large Scale Multilingual Multitask Benchmark for Code Understanding, Generation, Translation and Retrieval","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Md Rizwan Parvez, Mohammad Abdullah Matin Khan, M Saiful Bari, Shafiq Joty, Weishi Wang, Xuan Long Do","submitted_at":"2023-03-06T10:08:51Z","abstract_excerpt":"Recently, pre-trained large language models (LLMs) have shown impressive abilities in generating codes from natural language descriptions, repairing buggy codes, translating codes between languages, and retrieving relevant code segments. However, the evaluation of these models has often been performed in a scattered way on only one or two specific tasks, in a few languages, at a partial granularity (e.g., function) level, and in many cases without proper training data. Even more concerning is that in most cases the evaluation of generated codes has been done in terms of mere lexical overlap wi"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2303.03004","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2303.03004/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2303.03004","created_at":"2026-07-05T07:08:55.371057+00:00"},{"alias_kind":"arxiv_version","alias_value":"2303.03004v4","created_at":"2026-07-05T07:08:55.371057+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2303.03004","created_at":"2026-07-05T07:08:55.371057+00:00"},{"alias_kind":"pith_short_12","alias_value":"W6PC23N772VY","created_at":"2026-07-05T07:08:55.371057+00:00"},{"alias_kind":"pith_short_16","alias_value":"W6PC23N772VYILUE","created_at":"2026-07-05T07:08:55.371057+00:00"},{"alias_kind":"pith_short_8","alias_value":"W6PC23N7","created_at":"2026-07-05T07:08:55.371057+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":11,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2604.19741","citing_title":"CityRAG: Stepping Into a City via Spatially-Grounded Video Generation","ref_index":30,"is_internal_anchor":true},{"citing_arxiv_id":"2606.20517","citing_title":"Multi-LCB: Extending LiveCodeBench to Multiple Programming Languages","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2606.04273","citing_title":"Human agency in initial human-AI proof formalization workflows","ref_index":42,"is_internal_anchor":false},{"citing_arxiv_id":"2509.23108","citing_title":"Artificial Phantasia: Emergent Mental Imagery in Large Language Models","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2604.16321","citing_title":"LLM-Based Multi-Agent Systems for Code Generation: A Multi-Vocal Literature Review","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2406.00515","citing_title":"A Survey on Large Language Models for Code Generation","ref_index":131,"is_internal_anchor":false},{"citing_arxiv_id":"2605.04894","citing_title":"SynConfRoute: Syntax-Aware Routing for Efficient Code Completion with Small CodeLLMs","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2604.19742","citing_title":"PlayCoder: Making LLM-Generated GUI Code Playable","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07248","citing_title":"PaT: Planning-after-Trial for Efficient Test-Time Code Generation","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2403.07974","citing_title":"LiveCodeBench: Holistic and Contamination Free Evaluation of Large Language Models for Code","ref_index":264,"is_internal_anchor":false},{"citing_arxiv_id":"2604.17016","citing_title":"HELO-APR: Enhancing Low-Resource Program Repair through Cross-Lingual Knowledge Transfer","ref_index":14,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/W6PC23N772VYILUEPQULH4MNZG","json":"https://pith.science/pith/W6PC23N772VYILUEPQULH4MNZG.json","graph_json":"https://pith.science/api/pith-number/W6PC23N772VYILUEPQULH4MNZG/graph.json","events_json":"https://pith.science/api/pith-number/W6PC23N772VYILUEPQULH4MNZG/events.json","paper":"https://pith.science/paper/W6PC23N7"},"agent_actions":{"view_html":"https://pith.science/pith/W6PC23N772VYILUEPQULH4MNZG","download_json":"https://pith.science/pith/W6PC23N772VYILUEPQULH4MNZG.json","view_paper":"https://pith.science/paper/W6PC23N7","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2303.03004&json=true","fetch_graph":"https://pith.science/api/pith-number/W6PC23N772VYILUEPQULH4MNZG/graph.json","fetch_events":"https://pith.science/api/pith-number/W6PC23N772VYILUEPQULH4MNZG/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/W6PC23N772VYILUEPQULH4MNZG/action/timestamp_anchor","attest_storage":"https://pith.science/pith/W6PC23N772VYILUEPQULH4MNZG/action/storage_attestation","attest_author":"https://pith.science/pith/W6PC23N772VYILUEPQULH4MNZG/action/author_attestation","sign_citation":"https://pith.science/pith/W6PC23N772VYILUEPQULH4MNZG/action/citation_signature","submit_replication":"https://pith.science/pith/W6PC23N772VYILUEPQULH4MNZG/action/replication_record"}},"created_at":"2026-07-05T07:08:55.371057+00:00","updated_at":"2026-07-05T07:08:55.371057+00:00"}