{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:KACLCFGEBZ47ZS6GE5HRGEYCVB","short_pith_number":"pith:KACLCFGE","schema_version":"1.0","canonical_sha256":"5004b114c40e79fccbc6274f131302a84e1bf6a4a737a33c6841a3f93af1bf6c","source":{"kind":"arxiv","id":"2406.11409","version":2},"attestation_state":"computed","paper":{"title":"CodeGemma: Open Code Models Based on Gemma","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Ale Jakse Hartman, Andrea Hu, Bin Ni, Christopher A. Choquette-Choo, CodeGemma Team: Heri Zhao, Jane Fine, Jeffrey Hui, Jingyue Shen, Joe Kelley, Joshua Howland, Kathy Korevec, Kelly Schaefer, Kshitij Bansal, Luke Vilnis, Mateo Wirth, Nam Nguyen, Paul Michel, Peter Choy, Pratik Joshi, Ravin Kumar, Sarmad Hashmi, Scott Huffman, Shubham Agrawal, Siqi Zuo, Tris Warkentin, Zhitao Gong","submitted_at":"2024-06-17T10:54:35Z","abstract_excerpt":"This paper introduces CodeGemma, a collection of specialized open code models built on top of Gemma, capable of a variety of code and natural language generation tasks. We release three model variants. CodeGemma 7B pretrained (PT) and instruction-tuned (IT) variants have remarkably resilient natural language understanding, excel in mathematical reasoning, and match code capabilities of other open models. CodeGemma 2B is a state-of-the-art code completion model designed for fast code infilling and open-ended generation in latency-sensitive settings."},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2406.11409","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-06-17T10:54:35Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"412def6563faa278c9096fadab5fd8a4285dca8fa2476720906dc6fae0bf4588","abstract_canon_sha256":"4dcbecbf6e486108564a4de6cb219288bace83bbc876bc3f4ca8dd79d7f41a11"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:34:21.706427Z","signature_b64":"P5SQMDCw3gwQO92SUaOB/MJyw5YXF+MOFPR5ZEng8CIA/LnZL0c37BDG/QzH0bHEjJg8X+Je2tM5mfwE2GwjAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"5004b114c40e79fccbc6274f131302a84e1bf6a4a737a33c6841a3f93af1bf6c","last_reissued_at":"2026-07-05T08:34:21.705964Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:34:21.705964Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"CodeGemma: Open Code Models Based on Gemma","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Ale Jakse Hartman, Andrea Hu, Bin Ni, Christopher A. Choquette-Choo, CodeGemma Team: Heri Zhao, Jane Fine, Jeffrey Hui, Jingyue Shen, Joe Kelley, Joshua Howland, Kathy Korevec, Kelly Schaefer, Kshitij Bansal, Luke Vilnis, Mateo Wirth, Nam Nguyen, Paul Michel, Peter Choy, Pratik Joshi, Ravin Kumar, Sarmad Hashmi, Scott Huffman, Shubham Agrawal, Siqi Zuo, Tris Warkentin, Zhitao Gong","submitted_at":"2024-06-17T10:54:35Z","abstract_excerpt":"This paper introduces CodeGemma, a collection of specialized open code models built on top of Gemma, capable of a variety of code and natural language generation tasks. We release three model variants. CodeGemma 7B pretrained (PT) and instruction-tuned (IT) variants have remarkably resilient natural language understanding, excel in mathematical reasoning, and match code capabilities of other open models. CodeGemma 2B is a state-of-the-art code completion model designed for fast code infilling and open-ended generation in latency-sensitive settings."},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.11409","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.11409/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2406.11409","created_at":"2026-07-05T08:34:21.706023+00:00"},{"alias_kind":"arxiv_version","alias_value":"2406.11409v2","created_at":"2026-07-05T08:34:21.706023+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.11409","created_at":"2026-07-05T08:34:21.706023+00:00"},{"alias_kind":"pith_short_12","alias_value":"KACLCFGEBZ47","created_at":"2026-07-05T08:34:21.706023+00:00"},{"alias_kind":"pith_short_16","alias_value":"KACLCFGEBZ47ZS6G","created_at":"2026-07-05T08:34:21.706023+00:00"},{"alias_kind":"pith_short_8","alias_value":"KACLCFGE","created_at":"2026-07-05T08:34:21.706023+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":19,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.07881","citing_title":"Functional and Secure Code Generation with Task Vectors","ref_index":29,"is_internal_anchor":true},{"citing_arxiv_id":"2606.24901","citing_title":"LLM Evolution as an Industry-Scale Ecosystem: A Lifecycle Perspective on Continual Learning","ref_index":96,"is_internal_anchor":false},{"citing_arxiv_id":"2607.02186","citing_title":"UA-ChatDev: Uncertainty-Aware Multi-Agent Collaboration for Reliable Software Development","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2606.11755","citing_title":"Acoda: Adversarial Code Obfuscation for Defending against LLM-based Analysis","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2606.09145","citing_title":"PrivCode++: Latent-Conditioned Differentially Private Code Generation for Comprehensive Guarantees","ref_index":52,"is_internal_anchor":false},{"citing_arxiv_id":"2606.07999","citing_title":"Efficient Skill Grounding via Code Refactoring with Small Language Models","ref_index":75,"is_internal_anchor":false},{"citing_arxiv_id":"2606.03128","citing_title":"Decoupled Smart Contract Audits: Lightweight LLM Framework via Distillation and Aggregation","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2606.29815","citing_title":"SrDetection: A Self-Referential Framework for Data Leakage Detection in Code Large Language Models","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2605.25296","citing_title":"Subjective Code Preferences in Experts and Large Language Models","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2509.03117","citing_title":"PromptCOS: Towards Content-only System Prompt Copyright Auditing for LLMs","ref_index":50,"is_internal_anchor":false},{"citing_arxiv_id":"2410.22240","citing_title":"Are Decoder-Only Large Language Models the Silver Bullet for Code Search?","ref_index":64,"is_internal_anchor":false},{"citing_arxiv_id":"2502.06556","citing_title":"MultiFileTest: A Multi-File-Level LLM Unit Test Generation Benchmark and Impact of Error Fixing Mechanisms","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2505.10443","citing_title":"Are Large Language Models Robust in Understanding Code Against Semantics-Preserving Mutations?","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2507.06261","citing_title":"Gemini 2.5: Pushing the Frontier with Advanced Reasoning, Multimodality, Long Context, and Next Generation Agentic Capabilities","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2509.01082","citing_title":"RefineStat: Efficient Exploration for Probabilistic Program Synthesis","ref_index":46,"is_internal_anchor":false},{"citing_arxiv_id":"2409.12917","citing_title":"Training Language Models to Self-Correct via Reinforcement Learning","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2605.04894","citing_title":"SynConfRoute: Syntax-Aware Routing for Efficient Code Completion with Small CodeLLMs","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2604.19826","citing_title":"Co-Located Tests, Better AI Code: How Test Syntax Structure Affects Foundation Model Code Generation","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2604.21365","citing_title":"mcdok at SemEval-2026 Task 13: Finetuning LLMs for Detection of Machine-Generated Code","ref_index":7,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/KACLCFGEBZ47ZS6GE5HRGEYCVB","json":"https://pith.science/pith/KACLCFGEBZ47ZS6GE5HRGEYCVB.json","graph_json":"https://pith.science/api/pith-number/KACLCFGEBZ47ZS6GE5HRGEYCVB/graph.json","events_json":"https://pith.science/api/pith-number/KACLCFGEBZ47ZS6GE5HRGEYCVB/events.json","paper":"https://pith.science/paper/KACLCFGE"},"agent_actions":{"view_html":"https://pith.science/pith/KACLCFGEBZ47ZS6GE5HRGEYCVB","download_json":"https://pith.science/pith/KACLCFGEBZ47ZS6GE5HRGEYCVB.json","view_paper":"https://pith.science/paper/KACLCFGE","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2406.11409&json=true","fetch_graph":"https://pith.science/api/pith-number/KACLCFGEBZ47ZS6GE5HRGEYCVB/graph.json","fetch_events":"https://pith.science/api/pith-number/KACLCFGEBZ47ZS6GE5HRGEYCVB/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/KACLCFGEBZ47ZS6GE5HRGEYCVB/action/timestamp_anchor","attest_storage":"https://pith.science/pith/KACLCFGEBZ47ZS6GE5HRGEYCVB/action/storage_attestation","attest_author":"https://pith.science/pith/KACLCFGEBZ47ZS6GE5HRGEYCVB/action/author_attestation","sign_citation":"https://pith.science/pith/KACLCFGEBZ47ZS6GE5HRGEYCVB/action/citation_signature","submit_replication":"https://pith.science/pith/KACLCFGEBZ47ZS6GE5HRGEYCVB/action/replication_record"}},"created_at":"2026-07-05T08:34:21.706023+00:00","updated_at":"2026-07-05T08:34:21.706023+00:00"}