{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:EBCHHPZXHRJT4OMNIP437BYQ5K","short_pith_number":"pith:EBCHHPZX","schema_version":"1.0","canonical_sha256":"204473bf373c533e398d43f9bf8710ea8ca4ec1ee0c8da61aff2e15c77a99928","source":{"kind":"arxiv","id":"2411.06145","version":4},"attestation_state":"computed","paper":{"title":"ClassEval-T: Evaluating Large Language Models in Class-Level Code Translation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.SE","authors_text":"Chengyi Wang, Jacky Wai Keung, Jia Li, Linhao Wu, Pengyu Xue, Ruikai Jin, Xiang Li, Xiran Lyu, Yifei Pei, Yuxiang Zhang, Zhaoyan Shen, Zhen Yang","submitted_at":"2024-11-09T11:13:14Z","abstract_excerpt":"In recent years, Large Language Models (LLMs) have dramatically advanced the performance of automated code translation, making their computational accuracy score reach up to over 80% on many previous benchmarks. However, most code samples in these benchmarks are short, standalone, statement/method-level, and algorithmic, which is not aligned with practical coding tasks. Therefore, it is still unknown the actual capability of LLMs in translating code samples written for daily development. To achieve this, we construct a class-level code translation benchmark, ClassEval-T, and make the first att"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2411.06145","kind":"arxiv","version":4},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.SE","submitted_at":"2024-11-09T11:13:14Z","cross_cats_sorted":[],"title_canon_sha256":"664b871063ee4cf1f05cb239700a2ad81f43c9e184780ebd675f8100b73b2346","abstract_canon_sha256":"3fa88ed12d0d371d776e175f2b316e8211d24180a26dff963bad231a41fe7fdc"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:48:21.267647Z","signature_b64":"4omUCvFvf4d5Sy1Kmz2rkFrmg6TUAxrGQ0x409rAfgzp5czNe/j8ZwbVquRZMf9tgfz/rORi7OkjBFXJiZjLAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"204473bf373c533e398d43f9bf8710ea8ca4ec1ee0c8da61aff2e15c77a99928","last_reissued_at":"2026-07-05T10:48:21.267147Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:48:21.267147Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"ClassEval-T: Evaluating Large Language Models in Class-Level Code Translation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.SE","authors_text":"Chengyi Wang, Jacky Wai Keung, Jia Li, Linhao Wu, Pengyu Xue, Ruikai Jin, Xiang Li, Xiran Lyu, Yifei Pei, Yuxiang Zhang, Zhaoyan Shen, Zhen Yang","submitted_at":"2024-11-09T11:13:14Z","abstract_excerpt":"In recent years, Large Language Models (LLMs) have dramatically advanced the performance of automated code translation, making their computational accuracy score reach up to over 80% on many previous benchmarks. However, most code samples in these benchmarks are short, standalone, statement/method-level, and algorithmic, which is not aligned with practical coding tasks. Therefore, it is still unknown the actual capability of LLMs in translating code samples written for daily development. To achieve this, we construct a class-level code translation benchmark, ClassEval-T, and make the first att"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2411.06145","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2411.06145/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2411.06145","created_at":"2026-07-05T10:48:21.267203+00:00"},{"alias_kind":"arxiv_version","alias_value":"2411.06145v4","created_at":"2026-07-05T10:48:21.267203+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2411.06145","created_at":"2026-07-05T10:48:21.267203+00:00"},{"alias_kind":"pith_short_12","alias_value":"EBCHHPZXHRJT","created_at":"2026-07-05T10:48:21.267203+00:00"},{"alias_kind":"pith_short_16","alias_value":"EBCHHPZXHRJT4OMN","created_at":"2026-07-05T10:48:21.267203+00:00"},{"alias_kind":"pith_short_8","alias_value":"EBCHHPZX","created_at":"2026-07-05T10:48:21.267203+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.05249","citing_title":"SWE-InfraBench: Evaluating Language Models on Cloud Infrastructure Code","ref_index":15,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/EBCHHPZXHRJT4OMNIP437BYQ5K","json":"https://pith.science/pith/EBCHHPZXHRJT4OMNIP437BYQ5K.json","graph_json":"https://pith.science/api/pith-number/EBCHHPZXHRJT4OMNIP437BYQ5K/graph.json","events_json":"https://pith.science/api/pith-number/EBCHHPZXHRJT4OMNIP437BYQ5K/events.json","paper":"https://pith.science/paper/EBCHHPZX"},"agent_actions":{"view_html":"https://pith.science/pith/EBCHHPZXHRJT4OMNIP437BYQ5K","download_json":"https://pith.science/pith/EBCHHPZXHRJT4OMNIP437BYQ5K.json","view_paper":"https://pith.science/paper/EBCHHPZX","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2411.06145&json=true","fetch_graph":"https://pith.science/api/pith-number/EBCHHPZXHRJT4OMNIP437BYQ5K/graph.json","fetch_events":"https://pith.science/api/pith-number/EBCHHPZXHRJT4OMNIP437BYQ5K/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/EBCHHPZXHRJT4OMNIP437BYQ5K/action/timestamp_anchor","attest_storage":"https://pith.science/pith/EBCHHPZXHRJT4OMNIP437BYQ5K/action/storage_attestation","attest_author":"https://pith.science/pith/EBCHHPZXHRJT4OMNIP437BYQ5K/action/author_attestation","sign_citation":"https://pith.science/pith/EBCHHPZXHRJT4OMNIP437BYQ5K/action/citation_signature","submit_replication":"https://pith.science/pith/EBCHHPZXHRJT4OMNIP437BYQ5K/action/replication_record"}},"created_at":"2026-07-05T10:48:21.267203+00:00","updated_at":"2026-07-05T10:48:21.267203+00:00"}