{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:3J4VEV6AJ2PLWZ3D3EEFA574KJ","short_pith_number":"pith:3J4VEV6A","schema_version":"1.0","canonical_sha256":"da795257c04e9ebb6763d9085077fc527011c813d25e99b4451887ce3ae12a23","source":{"kind":"arxiv","id":"2312.02120","version":2},"attestation_state":"computed","paper":{"title":"Magicoder: Empowering Code Generation with OSS-Instruct","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.SE"],"primary_cat":"cs.CL","authors_text":"Jiawei Liu, Lingming Zhang, Yifeng Ding, Yuxiang Wei, Zhe Wang","submitted_at":"2023-12-04T18:50:35Z","abstract_excerpt":"We introduce Magicoder, a series of fully open-source (code, weights, and data) Large Language Models (LLMs) for code that significantly closes the gap with top code models while having no more than 7B parameters. Magicoder models are trained on 75K synthetic instruction data using OSS-Instruct, a novel approach to enlightening LLMs with open-source code snippets to generate diverse instruction data for code. Our main motivation is to mitigate the inherent bias of the synthetic data generated by LLMs through the wealth of open-source references for the production of more realistic and controll"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2312.02120","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2023-12-04T18:50:35Z","cross_cats_sorted":["cs.AI","cs.SE"],"title_canon_sha256":"7673c4a8e8139581b7b112b41302d9e74496da843a67162131712108bfbb006b","abstract_canon_sha256":"a2aadca9c7b1a0fd3fa9d52e185865491f61c5b86a7c3d6e942f8d1a4988df82"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:28:35.800450Z","signature_b64":"YlywgVrPp31vSx9mmV6EIkj/wKQ4l8huA6UwemkzEfR7yDhG68d6yVu5zXXfVJuvgApsduN7GIxeuXR60RNvBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"da795257c04e9ebb6763d9085077fc527011c813d25e99b4451887ce3ae12a23","last_reissued_at":"2026-07-05T08:28:35.799966Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:28:35.799966Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Magicoder: Empowering Code Generation with OSS-Instruct","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.SE"],"primary_cat":"cs.CL","authors_text":"Jiawei Liu, Lingming Zhang, Yifeng Ding, Yuxiang Wei, Zhe Wang","submitted_at":"2023-12-04T18:50:35Z","abstract_excerpt":"We introduce Magicoder, a series of fully open-source (code, weights, and data) Large Language Models (LLMs) for code that significantly closes the gap with top code models while having no more than 7B parameters. Magicoder models are trained on 75K synthetic instruction data using OSS-Instruct, a novel approach to enlightening LLMs with open-source code snippets to generate diverse instruction data for code. Our main motivation is to mitigate the inherent bias of the synthetic data generated by LLMs through the wealth of open-source references for the production of more realistic and controll"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2312.02120","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2312.02120/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2312.02120","created_at":"2026-07-05T08:28:35.800028+00:00"},{"alias_kind":"arxiv_version","alias_value":"2312.02120v2","created_at":"2026-07-05T08:28:35.800028+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2312.02120","created_at":"2026-07-05T08:28:35.800028+00:00"},{"alias_kind":"pith_short_12","alias_value":"3J4VEV6AJ2PL","created_at":"2026-07-05T08:28:35.800028+00:00"},{"alias_kind":"pith_short_16","alias_value":"3J4VEV6AJ2PLWZ3D","created_at":"2026-07-05T08:28:35.800028+00:00"},{"alias_kind":"pith_short_8","alias_value":"3J4VEV6A","created_at":"2026-07-05T08:28:35.800028+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":32,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.18286","citing_title":"CODEBLOCK: Learning to Supervise Code at the Right Granularity","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2606.07925","citing_title":"ROSUM-MCTS: Monte Carlo Tree Search-Inspired HDL Code Summarization with Structural Rewards","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2606.06826","citing_title":"SkelDPO: A Skeleton-Guided Direct Preference Optimization Framework for Efficient Code Generation","ref_index":47,"is_internal_anchor":false},{"citing_arxiv_id":"2606.03344","citing_title":"RogueMerge: Robust and Unified Attacks against LLM Model Merging","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2606.01286","citing_title":"BenchEvolver: Frontier Task Synthesis via Solution-Centric Evolution","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2605.24079","citing_title":"TRACER: A Semantic-Aware Framework for Fine-Grained Contamination Detection in Code LLMs","ref_index":41,"is_internal_anchor":false},{"citing_arxiv_id":"2606.28998","citing_title":"Reward-Free Code Alignment from Pretrained or Fine-Tuned LLM: Unpacking the Trade-offs for Code Generation","ref_index":54,"is_internal_anchor":false},{"citing_arxiv_id":"2605.25835","citing_title":"Context-Instrumental Data Distillation for Kubernetes Manifest Generation: Method and Experimental Evaluation","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2606.27601","citing_title":"Test Case Selection for Deep Neural Networks: A Replication Study on LLMs for Code","ref_index":68,"is_internal_anchor":false},{"citing_arxiv_id":"2606.08676","citing_title":"Lost in the Flow with Code Talkers: Unveiling the Instruction-Tuning Tax of Large Language Models in Code Tasks","ref_index":47,"is_internal_anchor":false},{"citing_arxiv_id":"2606.19988","citing_title":"Repository-Level Solidity Code Generation with Large Language Models: From Prompting to Fine-Tuning","ref_index":69,"is_internal_anchor":false},{"citing_arxiv_id":"2408.01055","citing_title":"Towards Agentic Runtime Healing","ref_index":58,"is_internal_anchor":false},{"citing_arxiv_id":"2602.08167","citing_title":"Self-Supervised Bootstrapping of Action-Predictive Embodied Reasoning","ref_index":59,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18829","citing_title":"Lossless Anti-Distillation Sampling","ref_index":130,"is_internal_anchor":false},{"citing_arxiv_id":"2605.16462","citing_title":"Asking Back: Interaction-Layer Antidistillation Watermarks","ref_index":39,"is_internal_anchor":false},{"citing_arxiv_id":"2509.21892","citing_title":"Elastic MoE: Unlocking the Inference-Time Scalability of Mixture-of-Experts","ref_index":38,"is_internal_anchor":false},{"citing_arxiv_id":"2510.15079","citing_title":"Assessing Coherency and Consistency of Code Execution Reasoning by Large Language Models","ref_index":45,"is_internal_anchor":false},{"citing_arxiv_id":"2512.18470","citing_title":"SWE-EVO: Benchmarking Coding Agents in Long-Horizon Software Evolution Scenarios","ref_index":54,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12843","citing_title":"Bayesian Model Merging","ref_index":55,"is_internal_anchor":false},{"citing_arxiv_id":"2406.00515","citing_title":"A Survey on Large Language Models for Code Generation","ref_index":285,"is_internal_anchor":false},{"citing_arxiv_id":"2404.06395","citing_title":"MiniCPM: Unveiling the Potential of Small Language Models with Scalable Training Strategies","ref_index":41,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12264","citing_title":"Reconstruction of Personally Identifiable Information from Supervised Finetuned Models","ref_index":58,"is_internal_anchor":false},{"citing_arxiv_id":"2407.01489","citing_title":"Agentless: Demystifying LLM-based Software Engineering Agents","ref_index":93,"is_internal_anchor":false},{"citing_arxiv_id":"2604.25960","citing_title":"Large Language Models for Multilingual Code Intelligence: A Survey","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06654","citing_title":"Optimizer-Model Consistency: Full Finetuning with the Same Optimizer as Pretraining Forgets Less","ref_index":32,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/3J4VEV6AJ2PLWZ3D3EEFA574KJ","json":"https://pith.science/pith/3J4VEV6AJ2PLWZ3D3EEFA574KJ.json","graph_json":"https://pith.science/api/pith-number/3J4VEV6AJ2PLWZ3D3EEFA574KJ/graph.json","events_json":"https://pith.science/api/pith-number/3J4VEV6AJ2PLWZ3D3EEFA574KJ/events.json","paper":"https://pith.science/paper/3J4VEV6A"},"agent_actions":{"view_html":"https://pith.science/pith/3J4VEV6AJ2PLWZ3D3EEFA574KJ","download_json":"https://pith.science/pith/3J4VEV6AJ2PLWZ3D3EEFA574KJ.json","view_paper":"https://pith.science/paper/3J4VEV6A","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2312.02120&json=true","fetch_graph":"https://pith.science/api/pith-number/3J4VEV6AJ2PLWZ3D3EEFA574KJ/graph.json","fetch_events":"https://pith.science/api/pith-number/3J4VEV6AJ2PLWZ3D3EEFA574KJ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/3J4VEV6AJ2PLWZ3D3EEFA574KJ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/3J4VEV6AJ2PLWZ3D3EEFA574KJ/action/storage_attestation","attest_author":"https://pith.science/pith/3J4VEV6AJ2PLWZ3D3EEFA574KJ/action/author_attestation","sign_citation":"https://pith.science/pith/3J4VEV6AJ2PLWZ3D3EEFA574KJ/action/citation_signature","submit_replication":"https://pith.science/pith/3J4VEV6AJ2PLWZ3D3EEFA574KJ/action/replication_record"}},"created_at":"2026-07-05T08:28:35.800028+00:00","updated_at":"2026-07-05T08:28:35.800028+00:00"}