{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2025:DSIFNLWELHBYNXS72AD56UK5FP","short_pith_number":"pith:DSIFNLWE","canonical_record":{"source":{"id":"2505.05063","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-05-08T08:55:32Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"d7fea1bb1af29ab241661047857cde3def43b5cf2f9c0e79c2b543f68cc0fd0a","abstract_canon_sha256":"df4a551a04be12da7f4cb30019cbd76990d3667d0264c9c9c74f9258193a44a5"},"schema_version":"1.0"},"canonical_sha256":"1c9056aec459c386de5fd007df515d2bc85be8c021edb326660e6c7f4d5d4f3f","source":{"kind":"arxiv","id":"2505.05063","version":1},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2505.05063","created_at":"2026-07-05T11:00:18Z"},{"alias_kind":"arxiv_version","alias_value":"2505.05063v1","created_at":"2026-07-05T11:00:18Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.05063","created_at":"2026-07-05T11:00:18Z"},{"alias_kind":"pith_short_12","alias_value":"DSIFNLWELHBY","created_at":"2026-07-05T11:00:18Z"},{"alias_kind":"pith_short_16","alias_value":"DSIFNLWELHBYNXS7","created_at":"2026-07-05T11:00:18Z"},{"alias_kind":"pith_short_8","alias_value":"DSIFNLWE","created_at":"2026-07-05T11:00:18Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2025:DSIFNLWELHBYNXS72AD56UK5FP","target":"record","payload":{"canonical_record":{"source":{"id":"2505.05063","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-05-08T08:55:32Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"d7fea1bb1af29ab241661047857cde3def43b5cf2f9c0e79c2b543f68cc0fd0a","abstract_canon_sha256":"df4a551a04be12da7f4cb30019cbd76990d3667d0264c9c9c74f9258193a44a5"},"schema_version":"1.0"},"canonical_sha256":"1c9056aec459c386de5fd007df515d2bc85be8c021edb326660e6c7f4d5d4f3f","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:00:18.511177Z","signature_b64":"ypzSphgk14rCXpSfc0RlTJi/nN2Hg5Mo7hAEsgQTjxUFTDbcBMk3X7rVt+maCIsSYqTH3BGtaKqMgL3UkravCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"1c9056aec459c386de5fd007df515d2bc85be8c021edb326660e6c7f4d5d4f3f","last_reissued_at":"2026-07-05T11:00:18.510621Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:00:18.510621Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2505.05063","source_version":1,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T11:00:18Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"a185dT8+APLENBs7jL7Gn0ih9fXrN7Kz2210J3EfLCMp/m4A0KgosroyaXAYevt0fCppyoxgHVYxgtHcucS9Bg==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-17T19:49:02.099971Z"},"content_sha256":"3d6c58e680c281f24c97d9f7d8dc73a4d7908cb6e921b16c08b4b181eb9778c7","schema_version":"1.0","event_id":"sha256:3d6c58e680c281f24c97d9f7d8dc73a4d7908cb6e921b16c08b4b181eb9778c7"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2025:DSIFNLWELHBYNXS72AD56UK5FP","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"CodeMixBench: Evaluating Large Language Models on Code Generation with Code-Mixed Prompts","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.LG","authors_text":"Manik Sheokand, Parth Sawant","submitted_at":"2025-05-08T08:55:32Z","abstract_excerpt":"Large Language Models (LLMs) have achieved remarkable success in code generation tasks, powering various applications like code completion, debugging, and programming assistance. However, existing benchmarks such as HumanEval, MBPP, and BigCodeBench primarily evaluate LLMs on English-only prompts, overlooking the real-world scenario where multilingual developers often use code-mixed language while interacting with LLMs. To address this gap, we introduce CodeMixBench, a novel benchmark designed to evaluate the robustness of LLMs on code generation from code-mixed prompts. Built upon BigCodeBenc"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.05063","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.05063/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T11:00:18Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"3VtCIwdyMovwY3aow2kvLokf2145jNJ6xMoMdMzWA+QMJCouDiHzhW3Ktimf8k0nJzrRj7vpKiQk4rgizMSuCw==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-17T19:49:02.100371Z"},"content_sha256":"301ba8c5b8b044c964c31f257fc59a81aeb63940aa1c3ab7cccef83bce4d0010","schema_version":"1.0","event_id":"sha256:301ba8c5b8b044c964c31f257fc59a81aeb63940aa1c3ab7cccef83bce4d0010"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/DSIFNLWELHBYNXS72AD56UK5FP/bundle.json","state_url":"https://pith.science/pith/DSIFNLWELHBYNXS72AD56UK5FP/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/DSIFNLWELHBYNXS72AD56UK5FP/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-17T19:49:02Z","links":{"resolver":"https://pith.science/pith/DSIFNLWELHBYNXS72AD56UK5FP","bundle":"https://pith.science/pith/DSIFNLWELHBYNXS72AD56UK5FP/bundle.json","state":"https://pith.science/pith/DSIFNLWELHBYNXS72AD56UK5FP/state.json","well_known_bundle":"https://pith.science/.well-known/pith/DSIFNLWELHBYNXS72AD56UK5FP/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2025:DSIFNLWELHBYNXS72AD56UK5FP","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"df4a551a04be12da7f4cb30019cbd76990d3667d0264c9c9c74f9258193a44a5","cross_cats_sorted":["cs.CL"],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-05-08T08:55:32Z","title_canon_sha256":"d7fea1bb1af29ab241661047857cde3def43b5cf2f9c0e79c2b543f68cc0fd0a"},"schema_version":"1.0","source":{"id":"2505.05063","kind":"arxiv","version":1}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2505.05063","created_at":"2026-07-05T11:00:18Z"},{"alias_kind":"arxiv_version","alias_value":"2505.05063v1","created_at":"2026-07-05T11:00:18Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.05063","created_at":"2026-07-05T11:00:18Z"},{"alias_kind":"pith_short_12","alias_value":"DSIFNLWELHBY","created_at":"2026-07-05T11:00:18Z"},{"alias_kind":"pith_short_16","alias_value":"DSIFNLWELHBYNXS7","created_at":"2026-07-05T11:00:18Z"},{"alias_kind":"pith_short_8","alias_value":"DSIFNLWE","created_at":"2026-07-05T11:00:18Z"}],"graph_snapshots":[{"event_id":"sha256:301ba8c5b8b044c964c31f257fc59a81aeb63940aa1c3ab7cccef83bce4d0010","target":"graph","created_at":"2026-07-05T11:00:18Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2505.05063/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"Large Language Models (LLMs) have achieved remarkable success in code generation tasks, powering various applications like code completion, debugging, and programming assistance. However, existing benchmarks such as HumanEval, MBPP, and BigCodeBench primarily evaluate LLMs on English-only prompts, overlooking the real-world scenario where multilingual developers often use code-mixed language while interacting with LLMs. To address this gap, we introduce CodeMixBench, a novel benchmark designed to evaluate the robustness of LLMs on code generation from code-mixed prompts. Built upon BigCodeBenc","authors_text":"Manik Sheokand, Parth Sawant","cross_cats":["cs.CL"],"headline":"","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-05-08T08:55:32Z","title":"CodeMixBench: Evaluating Large Language Models on Code Generation with Code-Mixed Prompts"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.05063","kind":"arxiv","version":1},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:3d6c58e680c281f24c97d9f7d8dc73a4d7908cb6e921b16c08b4b181eb9778c7","target":"record","created_at":"2026-07-05T11:00:18Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"df4a551a04be12da7f4cb30019cbd76990d3667d0264c9c9c74f9258193a44a5","cross_cats_sorted":["cs.CL"],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-05-08T08:55:32Z","title_canon_sha256":"d7fea1bb1af29ab241661047857cde3def43b5cf2f9c0e79c2b543f68cc0fd0a"},"schema_version":"1.0","source":{"id":"2505.05063","kind":"arxiv","version":1}},"canonical_sha256":"1c9056aec459c386de5fd007df515d2bc85be8c021edb326660e6c7f4d5d4f3f","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"1c9056aec459c386de5fd007df515d2bc85be8c021edb326660e6c7f4d5d4f3f","first_computed_at":"2026-07-05T11:00:18.510621Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T11:00:18.510621Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"ypzSphgk14rCXpSfc0RlTJi/nN2Hg5Mo7hAEsgQTjxUFTDbcBMk3X7rVt+maCIsSYqTH3BGtaKqMgL3UkravCA==","signature_status":"signed_v1","signed_at":"2026-07-05T11:00:18.511177Z","signed_message":"canonical_sha256_bytes"},"source_id":"2505.05063","source_kind":"arxiv","source_version":1}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:3d6c58e680c281f24c97d9f7d8dc73a4d7908cb6e921b16c08b4b181eb9778c7","sha256:301ba8c5b8b044c964c31f257fc59a81aeb63940aa1c3ab7cccef83bce4d0010"],"state_sha256":"d9b0b258c21e27f279c74fae447a209d9ae08035acd7cd530e62eedb4ecf6bb2"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"WsRTmcHNu11pl5llmKRI+AC9NJ2CIqNLg/3aeALMsHjsmEGrVlFOJNlpMzYCxsmPlp5+kwVYdtshU59reudWCg==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-17T19:49:02.103034Z","bundle_sha256":"418c315d4655e6146017d19e5203d5f2629a8f6c2f6206a82a96f247514024d1"}}