{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:ZOVG3XHBBWQZZVOND5VCZFFYW6","short_pith_number":"pith:ZOVG3XHB","schema_version":"1.0","canonical_sha256":"cbaa6ddce10da19cd5cd1f6a2c94b8b7b18526c655035fca1a6d72c884e1f92b","source":{"kind":"arxiv","id":"2206.08474","version":1},"attestation_state":"computed","paper":{"title":"XLCoST: A Benchmark Dataset for Cross-lingual Code Intelligence","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.SE","authors_text":"Aneesh Jain, Chandan K. Reddy, Karthik Suresh, Ming Zhu, Roshan Ravindran, Sindhu Tipirneni","submitted_at":"2022-06-16T22:49:39Z","abstract_excerpt":"Recent advances in machine learning have significantly improved the understanding of source code data and achieved good performance on a number of downstream tasks. Open source repositories like GitHub enable this process with rich unlabeled code data. However, the lack of high quality labeled data has largely hindered the progress of several code related tasks, such as program translation, summarization, synthesis, and code search. This paper introduces XLCoST, Cross-Lingual Code SnippeT dataset, a new benchmark dataset for cross-lingual code intelligence. Our dataset contains fine-grained pa"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2206.08474","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.SE","submitted_at":"2022-06-16T22:49:39Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"486fd24d00d1dd5937b822252ed7b82ed7fc449a8c7269669559af6c33ea8de8","abstract_canon_sha256":"f26c9482cc57d98cd21c78a1d97bed9f003dd2b8c69b7b6597a33bb7095c6e6a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T04:32:29.999383Z","signature_b64":"M6wYQBxAE9kMDDOSz7YwB0P9AoQXxlUf3wXeSLl9gexPIce64etyasFCzi5e56JPvgKschH5a/R+huM81Pm5Bw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"cbaa6ddce10da19cd5cd1f6a2c94b8b7b18526c655035fca1a6d72c884e1f92b","last_reissued_at":"2026-07-05T04:32:29.998902Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T04:32:29.998902Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"XLCoST: A Benchmark Dataset for Cross-lingual Code Intelligence","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.SE","authors_text":"Aneesh Jain, Chandan K. Reddy, Karthik Suresh, Ming Zhu, Roshan Ravindran, Sindhu Tipirneni","submitted_at":"2022-06-16T22:49:39Z","abstract_excerpt":"Recent advances in machine learning have significantly improved the understanding of source code data and achieved good performance on a number of downstream tasks. Open source repositories like GitHub enable this process with rich unlabeled code data. However, the lack of high quality labeled data has largely hindered the progress of several code related tasks, such as program translation, summarization, synthesis, and code search. This paper introduces XLCoST, Cross-Lingual Code SnippeT dataset, a new benchmark dataset for cross-lingual code intelligence. Our dataset contains fine-grained pa"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2206.08474","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2206.08474/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2206.08474","created_at":"2026-07-05T04:32:29.998957+00:00"},{"alias_kind":"arxiv_version","alias_value":"2206.08474v1","created_at":"2026-07-05T04:32:29.998957+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2206.08474","created_at":"2026-07-05T04:32:29.998957+00:00"},{"alias_kind":"pith_short_12","alias_value":"ZOVG3XHBBWQZ","created_at":"2026-07-05T04:32:29.998957+00:00"},{"alias_kind":"pith_short_16","alias_value":"ZOVG3XHBBWQZZVON","created_at":"2026-07-05T04:32:29.998957+00:00"},{"alias_kind":"pith_short_8","alias_value":"ZOVG3XHB","created_at":"2026-07-05T04:32:29.998957+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":9,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.11863","citing_title":"Enhancing LLM-Based Code Translation with Verified Multi-Semantic Representations","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2606.04273","citing_title":"Characterizing initial human-AI proof formalization workflows","ref_index":44,"is_internal_anchor":false},{"citing_arxiv_id":"2606.27747","citing_title":"UNICS: Multilingual Code Search via Unified Pseudocode and Contrastive Transfer Learning","ref_index":64,"is_internal_anchor":false},{"citing_arxiv_id":"2509.20881","citing_title":"PseudoBridge: Pseudo Code as the Bridge for Better Semantic and Logic Alignment in Code Retrieval","ref_index":58,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17978","citing_title":"AutoVecCoder: Teaching LLMs to Generate Explicitly Vectorized Code","ref_index":99,"is_internal_anchor":false},{"citing_arxiv_id":"2605.03356","citing_title":"POSTCONDBENCH: Benchmarking Correctness and Completeness in Formal Postcondition Inference","ref_index":206,"is_internal_anchor":false},{"citing_arxiv_id":"2604.25960","citing_title":"Large Language Models for Multilingual Code Intelligence: A Survey","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07788","citing_title":"Bridging the Programming Language Gap: Constructing a Multilingual Shared Semantic Space through AST Unification and Graph Matching","ref_index":39,"is_internal_anchor":false},{"citing_arxiv_id":"2403.07974","citing_title":"LiveCodeBench: Holistic and Contamination Free Evaluation of Large Language Models for Code","ref_index":17,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ZOVG3XHBBWQZZVOND5VCZFFYW6","json":"https://pith.science/pith/ZOVG3XHBBWQZZVOND5VCZFFYW6.json","graph_json":"https://pith.science/api/pith-number/ZOVG3XHBBWQZZVOND5VCZFFYW6/graph.json","events_json":"https://pith.science/api/pith-number/ZOVG3XHBBWQZZVOND5VCZFFYW6/events.json","paper":"https://pith.science/paper/ZOVG3XHB"},"agent_actions":{"view_html":"https://pith.science/pith/ZOVG3XHBBWQZZVOND5VCZFFYW6","download_json":"https://pith.science/pith/ZOVG3XHBBWQZZVOND5VCZFFYW6.json","view_paper":"https://pith.science/paper/ZOVG3XHB","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2206.08474&json=true","fetch_graph":"https://pith.science/api/pith-number/ZOVG3XHBBWQZZVOND5VCZFFYW6/graph.json","fetch_events":"https://pith.science/api/pith-number/ZOVG3XHBBWQZZVOND5VCZFFYW6/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ZOVG3XHBBWQZZVOND5VCZFFYW6/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ZOVG3XHBBWQZZVOND5VCZFFYW6/action/storage_attestation","attest_author":"https://pith.science/pith/ZOVG3XHBBWQZZVOND5VCZFFYW6/action/author_attestation","sign_citation":"https://pith.science/pith/ZOVG3XHBBWQZZVOND5VCZFFYW6/action/citation_signature","submit_replication":"https://pith.science/pith/ZOVG3XHBBWQZZVOND5VCZFFYW6/action/replication_record"}},"created_at":"2026-07-05T04:32:29.998957+00:00","updated_at":"2026-07-05T04:32:29.998957+00:00"}