{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:2CF3HZ2QURRID6LJBXLNIFKDVD","short_pith_number":"pith:2CF3HZ2Q","schema_version":"1.0","canonical_sha256":"d08bb3e750a46281f9690dd6d41543a8f9d5c307f5efe1174b085b304ba6cba4","source":{"kind":"arxiv","id":"2308.03109","version":3},"attestation_state":"computed","paper":{"title":"Lost in Translation: A Study of Bugs Introduced by Large Language Models while Translating Code","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.SE","authors_text":"Ali Reza Ibrahimzada, Boris Sobolev, Divya Sankar, Lambert Pouguem Wassi, Michele Merler, Rahul Krishna, Raju Pavuluri, Rangeet Pan, Reyhaneh Jabbarvand, Saurabh Sinha","submitted_at":"2023-08-06T13:33:13Z","abstract_excerpt":"Code translation aims to convert source code from one programming language (PL) to another. Given the promising abilities of large language models (LLMs) in code synthesis, researchers are exploring their potential to automate code translation. The prerequisite for advancing the state of LLM-based code translation is to understand their promises and limitations over existing techniques. To that end, we present a large-scale empirical study to investigate the ability of general LLMs and code LLMs for code translation across pairs of different languages, including C, C++, Go, Java, and Python. O"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2308.03109","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by-sa/4.0/","primary_cat":"cs.SE","submitted_at":"2023-08-06T13:33:13Z","cross_cats_sorted":[],"title_canon_sha256":"25b8bdc71990cadc4427debda3a17db3d927c4eb3d348d57399915b1a536cb0c","abstract_canon_sha256":"b1a566b80f4e553e733524705020bc11b4e177cf113e9d8bbb2e669a82fbc3c9"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:33:53.789892Z","signature_b64":"SHKhBqczVj/gyYSMVZIDxjYoayQtrzUEe1sXL8IMgEPDbM5UP3v9Xd7KXxC3qYBZfAYT6bvLwmgDKhRRnOUICQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"d08bb3e750a46281f9690dd6d41543a8f9d5c307f5efe1174b085b304ba6cba4","last_reissued_at":"2026-07-05T07:33:53.789398Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:33:53.789398Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Lost in Translation: A Study of Bugs Introduced by Large Language Models while Translating Code","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.SE","authors_text":"Ali Reza Ibrahimzada, Boris Sobolev, Divya Sankar, Lambert Pouguem Wassi, Michele Merler, Rahul Krishna, Raju Pavuluri, Rangeet Pan, Reyhaneh Jabbarvand, Saurabh Sinha","submitted_at":"2023-08-06T13:33:13Z","abstract_excerpt":"Code translation aims to convert source code from one programming language (PL) to another. Given the promising abilities of large language models (LLMs) in code synthesis, researchers are exploring their potential to automate code translation. The prerequisite for advancing the state of LLM-based code translation is to understand their promises and limitations over existing techniques. To that end, we present a large-scale empirical study to investigate the ability of general LLMs and code LLMs for code translation across pairs of different languages, including C, C++, Go, Java, and Python. O"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2308.03109","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2308.03109/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2308.03109","created_at":"2026-07-05T07:33:53.789457+00:00"},{"alias_kind":"arxiv_version","alias_value":"2308.03109v3","created_at":"2026-07-05T07:33:53.789457+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2308.03109","created_at":"2026-07-05T07:33:53.789457+00:00"},{"alias_kind":"pith_short_12","alias_value":"2CF3HZ2QURRI","created_at":"2026-07-05T07:33:53.789457+00:00"},{"alias_kind":"pith_short_16","alias_value":"2CF3HZ2QURRID6LJ","created_at":"2026-07-05T07:33:53.789457+00:00"},{"alias_kind":"pith_short_8","alias_value":"2CF3HZ2Q","created_at":"2026-07-05T07:33:53.789457+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2402.09664","citing_title":"CodeMind: Evaluating Large Language Models for Code Reasoning","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2601.02602","citing_title":"SWaRL: Safeguard Code Watermarking via Reinforcement Learning","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2605.13896","citing_title":"Neural Code Translation of Legacy Code: APL to C#","ref_index":8,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/2CF3HZ2QURRID6LJBXLNIFKDVD","json":"https://pith.science/pith/2CF3HZ2QURRID6LJBXLNIFKDVD.json","graph_json":"https://pith.science/api/pith-number/2CF3HZ2QURRID6LJBXLNIFKDVD/graph.json","events_json":"https://pith.science/api/pith-number/2CF3HZ2QURRID6LJBXLNIFKDVD/events.json","paper":"https://pith.science/paper/2CF3HZ2Q"},"agent_actions":{"view_html":"https://pith.science/pith/2CF3HZ2QURRID6LJBXLNIFKDVD","download_json":"https://pith.science/pith/2CF3HZ2QURRID6LJBXLNIFKDVD.json","view_paper":"https://pith.science/paper/2CF3HZ2Q","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2308.03109&json=true","fetch_graph":"https://pith.science/api/pith-number/2CF3HZ2QURRID6LJBXLNIFKDVD/graph.json","fetch_events":"https://pith.science/api/pith-number/2CF3HZ2QURRID6LJBXLNIFKDVD/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/2CF3HZ2QURRID6LJBXLNIFKDVD/action/timestamp_anchor","attest_storage":"https://pith.science/pith/2CF3HZ2QURRID6LJBXLNIFKDVD/action/storage_attestation","attest_author":"https://pith.science/pith/2CF3HZ2QURRID6LJBXLNIFKDVD/action/author_attestation","sign_citation":"https://pith.science/pith/2CF3HZ2QURRID6LJBXLNIFKDVD/action/citation_signature","submit_replication":"https://pith.science/pith/2CF3HZ2QURRID6LJBXLNIFKDVD/action/replication_record"}},"created_at":"2026-07-05T07:33:53.789457+00:00","updated_at":"2026-07-05T07:33:53.789457+00:00"}