{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2024:U5WBJQQ2DZVMI7O5D7SI2G5HIP","short_pith_number":"pith:U5WBJQQ2","canonical_record":{"source":{"id":"2410.15037","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.CL","submitted_at":"2024-10-19T08:44:26Z","cross_cats_sorted":[],"title_canon_sha256":"4558118d22a7a42504fc69172574cd3ec3ac88c89a75b53400396ca7752ac61e","abstract_canon_sha256":"f9fdbacdfa9e03da536fe1bbcccf227b65ab21a1e3f8764b1431c8b360aa3882"},"schema_version":"1.0"},"canonical_sha256":"a76c14c21a1e6ac47ddd1fe48d1ba743f469566fd027c8b7799d464b97c40609","source":{"kind":"arxiv","id":"2410.15037","version":2},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2410.15037","created_at":"2026-07-05T11:03:52Z"},{"alias_kind":"arxiv_version","alias_value":"2410.15037v2","created_at":"2026-07-05T11:03:52Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.15037","created_at":"2026-07-05T11:03:52Z"},{"alias_kind":"pith_short_12","alias_value":"U5WBJQQ2DZVM","created_at":"2026-07-05T11:03:52Z"},{"alias_kind":"pith_short_16","alias_value":"U5WBJQQ2DZVMI7O5","created_at":"2026-07-05T11:03:52Z"},{"alias_kind":"pith_short_8","alias_value":"U5WBJQQ2","created_at":"2026-07-05T11:03:52Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2024:U5WBJQQ2DZVMI7O5D7SI2G5HIP","target":"record","payload":{"canonical_record":{"source":{"id":"2410.15037","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.CL","submitted_at":"2024-10-19T08:44:26Z","cross_cats_sorted":[],"title_canon_sha256":"4558118d22a7a42504fc69172574cd3ec3ac88c89a75b53400396ca7752ac61e","abstract_canon_sha256":"f9fdbacdfa9e03da536fe1bbcccf227b65ab21a1e3f8764b1431c8b360aa3882"},"schema_version":"1.0"},"canonical_sha256":"a76c14c21a1e6ac47ddd1fe48d1ba743f469566fd027c8b7799d464b97c40609","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:03:52.904084Z","signature_b64":"uRKu20OMOZszPdDWceC2wiPwFsIL8X62k6RrYR9KWGyAlFguHBnJVdT+shDZ65jyaPoAv0hMLxzwneKw+TwbCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"a76c14c21a1e6ac47ddd1fe48d1ba743f469566fd027c8b7799d464b97c40609","last_reissued_at":"2026-07-05T11:03:52.903632Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:03:52.903632Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2410.15037","source_version":2,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T11:03:52Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"3ZSX/tat6aI+ti16wXV9dgDnzTvDvXyzS40xpxQ/FSnd4ZK6nel2r7T0See8QSynaqrmwgcJXbQMoBGeceC6Cg==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-09T06:58:06.608280Z"},"content_sha256":"08644c34e120657c18e121accc13c8213b6a5df7888accf849131abcc51d9d1c","schema_version":"1.0","event_id":"sha256:08644c34e120657c18e121accc13c8213b6a5df7888accf849131abcc51d9d1c"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2024:U5WBJQQ2DZVMI7O5D7SI2G5HIP","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"mHumanEval -- A Multilingual Benchmark to Evaluate Large Language Models for Code Generation","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Antonios Anastasopoulos, Marcos Zampieri, Nishat Raihan","submitted_at":"2024-10-19T08:44:26Z","abstract_excerpt":"Recent advancements in large language models (LLMs) have significantly enhanced code generation from natural language prompts. The HumanEval Benchmark, developed by OpenAI, remains the most widely used code generation benchmark. However, this and other Code LLM benchmarks face critical limitations, particularly in task diversity, test coverage, and linguistic scope. Current evaluations primarily focus on English-to-Python conversion tasks with limited test cases, potentially overestimating model performance. While recent works have addressed test coverage and programming language (PL) diversit"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.15037","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.15037/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T11:03:52Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"dvfcfN6aAe1N3cARnPVNUDrRQFLWGQC0WApcedDh5q76APUBLzQMAk4dIcjXq3fIp+jGYOiR4wcNGOVoHehpAA==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-09T06:58:06.609158Z"},"content_sha256":"108006fb31fdc4f133e13200165554db159aa8efe9d7f24f759bc54d04dc3cd0","schema_version":"1.0","event_id":"sha256:108006fb31fdc4f133e13200165554db159aa8efe9d7f24f759bc54d04dc3cd0"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/U5WBJQQ2DZVMI7O5D7SI2G5HIP/bundle.json","state_url":"https://pith.science/pith/U5WBJQQ2DZVMI7O5D7SI2G5HIP/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/U5WBJQQ2DZVMI7O5D7SI2G5HIP/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-09T06:58:06Z","links":{"resolver":"https://pith.science/pith/U5WBJQQ2DZVMI7O5D7SI2G5HIP","bundle":"https://pith.science/pith/U5WBJQQ2DZVMI7O5D7SI2G5HIP/bundle.json","state":"https://pith.science/pith/U5WBJQQ2DZVMI7O5D7SI2G5HIP/state.json","well_known_bundle":"https://pith.science/.well-known/pith/U5WBJQQ2DZVMI7O5D7SI2G5HIP/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2024:U5WBJQQ2DZVMI7O5D7SI2G5HIP","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"f9fdbacdfa9e03da536fe1bbcccf227b65ab21a1e3f8764b1431c8b360aa3882","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.CL","submitted_at":"2024-10-19T08:44:26Z","title_canon_sha256":"4558118d22a7a42504fc69172574cd3ec3ac88c89a75b53400396ca7752ac61e"},"schema_version":"1.0","source":{"id":"2410.15037","kind":"arxiv","version":2}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2410.15037","created_at":"2026-07-05T11:03:52Z"},{"alias_kind":"arxiv_version","alias_value":"2410.15037v2","created_at":"2026-07-05T11:03:52Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.15037","created_at":"2026-07-05T11:03:52Z"},{"alias_kind":"pith_short_12","alias_value":"U5WBJQQ2DZVM","created_at":"2026-07-05T11:03:52Z"},{"alias_kind":"pith_short_16","alias_value":"U5WBJQQ2DZVMI7O5","created_at":"2026-07-05T11:03:52Z"},{"alias_kind":"pith_short_8","alias_value":"U5WBJQQ2","created_at":"2026-07-05T11:03:52Z"}],"graph_snapshots":[{"event_id":"sha256:108006fb31fdc4f133e13200165554db159aa8efe9d7f24f759bc54d04dc3cd0","target":"graph","created_at":"2026-07-05T11:03:52Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2410.15037/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Recent advancements in large language models (LLMs) have significantly enhanced code generation from natural language prompts. The HumanEval Benchmark, developed by OpenAI, remains the most widely used code generation benchmark. However, this and other Code LLM benchmarks face critical limitations, particularly in task diversity, test coverage, and linguistic scope. Current evaluations primarily focus on English-to-Python conversion tasks with limited test cases, potentially overestimating model performance. While recent works have addressed test coverage and programming language (PL) diversit","authors_text":"Antonios Anastasopoulos, Marcos Zampieri, Nishat Raihan","cross_cats":[],"headline":"","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.CL","submitted_at":"2024-10-19T08:44:26Z","title":"mHumanEval -- A Multilingual Benchmark to Evaluate Large Language Models for Code Generation"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.15037","kind":"arxiv","version":2},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:08644c34e120657c18e121accc13c8213b6a5df7888accf849131abcc51d9d1c","target":"record","created_at":"2026-07-05T11:03:52Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"f9fdbacdfa9e03da536fe1bbcccf227b65ab21a1e3f8764b1431c8b360aa3882","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.CL","submitted_at":"2024-10-19T08:44:26Z","title_canon_sha256":"4558118d22a7a42504fc69172574cd3ec3ac88c89a75b53400396ca7752ac61e"},"schema_version":"1.0","source":{"id":"2410.15037","kind":"arxiv","version":2}},"canonical_sha256":"a76c14c21a1e6ac47ddd1fe48d1ba743f469566fd027c8b7799d464b97c40609","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"a76c14c21a1e6ac47ddd1fe48d1ba743f469566fd027c8b7799d464b97c40609","first_computed_at":"2026-07-05T11:03:52.903632Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T11:03:52.903632Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"uRKu20OMOZszPdDWceC2wiPwFsIL8X62k6RrYR9KWGyAlFguHBnJVdT+shDZ65jyaPoAv0hMLxzwneKw+TwbCw==","signature_status":"signed_v1","signed_at":"2026-07-05T11:03:52.904084Z","signed_message":"canonical_sha256_bytes"},"source_id":"2410.15037","source_kind":"arxiv","source_version":2}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:08644c34e120657c18e121accc13c8213b6a5df7888accf849131abcc51d9d1c","sha256:108006fb31fdc4f133e13200165554db159aa8efe9d7f24f759bc54d04dc3cd0"],"state_sha256":"681adc80844281d2989669b6d0647bfc5700eee6458f1eac81e9f9158a1c3cfb"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"EUWYeiWIfnL7dNNMpQxf0A43xqLaQqLiQFdxBZbki420gVF6DBkUyNSr+3OkJrFyykProAPeFv1CxARdHlWoAg==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-09T06:58:06.615424Z","bundle_sha256":"d38d5af4210304c84f731c694b74aed7ed42ddf5a127dec97031309c84a5f265"}}