{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2021:3ZSNA2BD7PMPV6O3GT7QHVCRD4","short_pith_number":"pith:3ZSNA2BD","schema_version":"1.0","canonical_sha256":"de64d06823fbd8faf9db34ff03d4511f18a790eadb343a8a35cfc6c6fb11eb2a","source":{"kind":"arxiv","id":"2105.12655","version":2},"attestation_state":"computed","paper":{"title":"CodeNet: A Large-Scale AI for Code Dataset for Learning a Diversity of Coding Tasks","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.SE","authors_text":"David S. Kung, Frederick Reiss, Geert Janssen, Giacomo Domeniconi, Jie Chen, Julian Dolby, Lindsey Decker, Luca Buratti, Mihir Choudhury, Ruchir Puri, Saurabh Pujar, Shyam Ramji, Susan Malaika, Ulrich Finkler, Veronika Thost, Vladimir Zolotov, Wei Zhang","submitted_at":"2021-05-25T00:13:29Z","abstract_excerpt":"Over the last several decades, software has been woven into the fabric of every aspect of our society. As software development surges and code infrastructure of enterprise applications ages, it is now more critical than ever to increase software development productivity and modernize legacy applications. Advances in deep learning and machine learning algorithms have enabled numerous breakthroughs, motivating researchers to leverage AI techniques to improve software development efficiency. Thus, the fast-emerging research area of AI for Code has garnered new interest and gathered momentum. In t"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2105.12655","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.SE","submitted_at":"2021-05-25T00:13:29Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"0228cc17102ac807dd19309a49dd65a1e46f6366b3b810df57b5347480fbf335","abstract_canon_sha256":"af1e4c95a701d80c1e0b8e495880f3747cfba6b6a2199c7b1760cc4abdaf2d23"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T03:09:30.425381Z","signature_b64":"V0wYny1ri2mILyaVT3nE8RUxR3sd/jFpzQw7fl+WZntieZM4lJnQBcBcrDwCnPHOTplR15BDOhktpI7PnVnCBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"de64d06823fbd8faf9db34ff03d4511f18a790eadb343a8a35cfc6c6fb11eb2a","last_reissued_at":"2026-07-05T03:09:30.424860Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T03:09:30.424860Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"CodeNet: A Large-Scale AI for Code Dataset for Learning a Diversity of Coding Tasks","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.SE","authors_text":"David S. Kung, Frederick Reiss, Geert Janssen, Giacomo Domeniconi, Jie Chen, Julian Dolby, Lindsey Decker, Luca Buratti, Mihir Choudhury, Ruchir Puri, Saurabh Pujar, Shyam Ramji, Susan Malaika, Ulrich Finkler, Veronika Thost, Vladimir Zolotov, Wei Zhang","submitted_at":"2021-05-25T00:13:29Z","abstract_excerpt":"Over the last several decades, software has been woven into the fabric of every aspect of our society. As software development surges and code infrastructure of enterprise applications ages, it is now more critical than ever to increase software development productivity and modernize legacy applications. Advances in deep learning and machine learning algorithms have enabled numerous breakthroughs, motivating researchers to leverage AI techniques to improve software development efficiency. Thus, the fast-emerging research area of AI for Code has garnered new interest and gathered momentum. In t"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2105.12655","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2105.12655/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2105.12655","created_at":"2026-07-05T03:09:30.424929+00:00"},{"alias_kind":"arxiv_version","alias_value":"2105.12655v2","created_at":"2026-07-05T03:09:30.424929+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2105.12655","created_at":"2026-07-05T03:09:30.424929+00:00"},{"alias_kind":"pith_short_12","alias_value":"3ZSNA2BD7PMP","created_at":"2026-07-05T03:09:30.424929+00:00"},{"alias_kind":"pith_short_16","alias_value":"3ZSNA2BD7PMPV6O3","created_at":"2026-07-05T03:09:30.424929+00:00"},{"alias_kind":"pith_short_8","alias_value":"3ZSNA2BD","created_at":"2026-07-05T03:09:30.424929+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":28,"internal_anchor_count":2,"sample":[{"citing_arxiv_id":"2607.07748","citing_title":"Selective Left-Shift: Turning Test-Time Compute and Difficulty-based Curation into Training Data for Low-Resource Code Generation","ref_index":20,"is_internal_anchor":true},{"citing_arxiv_id":"2607.06009","citing_title":"Multi-Channel Spread-Spectrum Code Watermarking","ref_index":23,"is_internal_anchor":true},{"citing_arxiv_id":"2606.17683","citing_title":"Bridging Functional Correctness and Runtime Efficiency Gaps in LLM-Based Code Translation","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2606.11755","citing_title":"Acoda: Adversarial Code Obfuscation for Defending against LLM-based Analysis","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2606.08840","citing_title":"Beyond Pass Rate: A Multilingual, Execution-Grounded Evaluation of Open Code LLMs","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2606.06821","citing_title":"Chiseling Out Efficiency: Structured Skeleton Supervision for Efficient Code Generation","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2606.01723","citing_title":"Shortcut to Nowhere: Demystifying Deep Spurious Regression","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2605.28409","citing_title":"Efficient Post-training of LLMs for Code Generation With Offline Reinforcement Learning","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2409.19894","citing_title":"TransAgent: Enhancing LLM-Based Code Translation via Fine-Grained Execution Alignment","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2412.04590","citing_title":"Specification-Driven Code Translation Powered by Large Language Models: How Far Are We?","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2412.14399","citing_title":"NESA: Relational Neuro-Symbolic Static Program Analysis","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2505.10708","citing_title":"SafeTrans: LLM-assisted Transpilation from C to Rust","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2603.16011","citing_title":"FormulaCode: Evaluating Agentic Optimization on Large Codebases","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2507.21954","citing_title":"Fine-Tuning Code Language Models to Detect Cross-Language Bugs","ref_index":55,"is_internal_anchor":false},{"citing_arxiv_id":"2604.02352","citing_title":"An Initial Exploration of Contrastive Prompt Tuning to Generate Energy-Efficient Code","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2605.13896","citing_title":"Neural Code Translation of Legacy Code: APL to C#","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09421","citing_title":"MACAA: Belief-Revision Multi-Agent Reasoning for Code Authorship Verification","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2112.00114","citing_title":"Show Your Work: Scratchpads for Intermediate Computation with Language Models","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09421","citing_title":"MACAA: Belief-Revision Multi-Agent Reasoning for Code Authorship Verification","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2604.25599","citing_title":"PLMGH: What Matters in PLM-GNN Hybrids for Code Classification and Vulnerability Detection","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2604.25960","citing_title":"Large Language Models for Multilingual Code Intelligence: A Survey","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2604.08083","citing_title":"Can LLMs Deobfuscate Binary Code? A Systematic Analysis of Large Language Models into Pseudocode Deobfuscation","ref_index":57,"is_internal_anchor":false},{"citing_arxiv_id":"2605.02195","citing_title":"Beyond Translation Accuracy: Addressing False Failures in LLM-Based Code Translation","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2303.17651","citing_title":"Self-Refine: Iterative Refinement with Self-Feedback","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2604.18027","citing_title":"CodePivot: Bootstrapping Multilingual Transpilation in LLMs via Reinforcement Learning without Parallel Corpora","ref_index":51,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/3ZSNA2BD7PMPV6O3GT7QHVCRD4","json":"https://pith.science/pith/3ZSNA2BD7PMPV6O3GT7QHVCRD4.json","graph_json":"https://pith.science/api/pith-number/3ZSNA2BD7PMPV6O3GT7QHVCRD4/graph.json","events_json":"https://pith.science/api/pith-number/3ZSNA2BD7PMPV6O3GT7QHVCRD4/events.json","paper":"https://pith.science/paper/3ZSNA2BD"},"agent_actions":{"view_html":"https://pith.science/pith/3ZSNA2BD7PMPV6O3GT7QHVCRD4","download_json":"https://pith.science/pith/3ZSNA2BD7PMPV6O3GT7QHVCRD4.json","view_paper":"https://pith.science/paper/3ZSNA2BD","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2105.12655&json=true","fetch_graph":"https://pith.science/api/pith-number/3ZSNA2BD7PMPV6O3GT7QHVCRD4/graph.json","fetch_events":"https://pith.science/api/pith-number/3ZSNA2BD7PMPV6O3GT7QHVCRD4/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/3ZSNA2BD7PMPV6O3GT7QHVCRD4/action/timestamp_anchor","attest_storage":"https://pith.science/pith/3ZSNA2BD7PMPV6O3GT7QHVCRD4/action/storage_attestation","attest_author":"https://pith.science/pith/3ZSNA2BD7PMPV6O3GT7QHVCRD4/action/author_attestation","sign_citation":"https://pith.science/pith/3ZSNA2BD7PMPV6O3GT7QHVCRD4/action/citation_signature","submit_replication":"https://pith.science/pith/3ZSNA2BD7PMPV6O3GT7QHVCRD4/action/replication_record"}},"created_at":"2026-07-05T03:09:30.424929+00:00","updated_at":"2026-07-05T03:09:30.424929+00:00"}