{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:OVAEHT7GC6T3QQHLEW5BFI3GRJ","short_pith_number":"pith:OVAEHT7G","schema_version":"1.0","canonical_sha256":"754043cfe617a7b840eb25ba12a3668a5756545dae10c2c56f61f1e1de6657b0","source":{"kind":"arxiv","id":"2308.07124","version":2},"attestation_state":"computed","paper":{"title":"OctoPack: Instruction Tuning Code Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Armel Zebaze, Binyuan Hui, Leandro Von Werra, Niklas Muennighoff, Qian Liu, Qinkai Zheng, Shayne Longpre, Swayam Singh, Terry Yue Zhuo, Xiangru Tang","submitted_at":"2023-08-14T13:53:54Z","abstract_excerpt":"Finetuning large language models (LLMs) on instructions leads to vast performance improvements on natural language tasks. We apply instruction tuning using code, leveraging the natural structure of Git commits, which pair code changes with human instructions. We compile CommitPack: 4 terabytes of Git commits across 350 programming languages. We benchmark CommitPack against other natural and synthetic code instructions (xP3x, Self-Instruct, OASST) on the 16B parameter StarCoder model, and achieve state-of-the-art performance among models not trained on OpenAI outputs, on the HumanEval Python be"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2308.07124","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2023-08-14T13:53:54Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"14218043273482b31d98f9ba54e9566891df036b1e7eb8ba94015dba2b108415","abstract_canon_sha256":"71aa540d49468c68a6e5e8ca4e2075bee2dedbef7a4543b072f73d129be12bd1"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:46:21.815574Z","signature_b64":"IPUzVExPyvrz/tYYKlj9LZ1hL6uzCPjeW64Y1h3W2w7WBbxxgzvtMjp2SYCQaI0+Rni4mxXrv/kCz2URsCj+DQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"754043cfe617a7b840eb25ba12a3668a5756545dae10c2c56f61f1e1de6657b0","last_reissued_at":"2026-07-05T07:46:21.815112Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:46:21.815112Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"OctoPack: Instruction Tuning Code Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Armel Zebaze, Binyuan Hui, Leandro Von Werra, Niklas Muennighoff, Qian Liu, Qinkai Zheng, Shayne Longpre, Swayam Singh, Terry Yue Zhuo, Xiangru Tang","submitted_at":"2023-08-14T13:53:54Z","abstract_excerpt":"Finetuning large language models (LLMs) on instructions leads to vast performance improvements on natural language tasks. We apply instruction tuning using code, leveraging the natural structure of Git commits, which pair code changes with human instructions. We compile CommitPack: 4 terabytes of Git commits across 350 programming languages. We benchmark CommitPack against other natural and synthetic code instructions (xP3x, Self-Instruct, OASST) on the 16B parameter StarCoder model, and achieve state-of-the-art performance among models not trained on OpenAI outputs, on the HumanEval Python be"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2308.07124","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2308.07124/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2308.07124","created_at":"2026-07-05T07:46:21.815169+00:00"},{"alias_kind":"arxiv_version","alias_value":"2308.07124v2","created_at":"2026-07-05T07:46:21.815169+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2308.07124","created_at":"2026-07-05T07:46:21.815169+00:00"},{"alias_kind":"pith_short_12","alias_value":"OVAEHT7GC6T3","created_at":"2026-07-05T07:46:21.815169+00:00"},{"alias_kind":"pith_short_16","alias_value":"OVAEHT7GC6T3QQHL","created_at":"2026-07-05T07:46:21.815169+00:00"},{"alias_kind":"pith_short_8","alias_value":"OVAEHT7G","created_at":"2026-07-05T07:46:21.815169+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":20,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.11470","citing_title":"The Periodic Table of LLM Reasoning: A Structured Survey of Reasoning Paradigms, Methods, and Failure Modes","ref_index":177,"is_internal_anchor":false},{"citing_arxiv_id":"2606.09659","citing_title":"End-to-End Context Compression at Scale","ref_index":62,"is_internal_anchor":false},{"citing_arxiv_id":"2606.28998","citing_title":"Reward-Free Code Alignment from Pretrained or Fine-Tuned LLM: Unpacking the Trade-offs for Code Generation","ref_index":36,"is_internal_anchor":false},{"citing_arxiv_id":"2402.09664","citing_title":"CodeMind: Evaluating Large Language Models for Code Reasoning","ref_index":40,"is_internal_anchor":false},{"citing_arxiv_id":"2505.13360","citing_title":"What Prompts Don't Say: Understanding and Managing Underspecification in LLM Prompts","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2509.24328","citing_title":"Speculative Verification: Exploiting Information Gain to Refine Speculative Decoding","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2510.01379","citing_title":"Multi-LLM Orchestration for High-Quality Code Generation: Exploiting Complementary Model Strengths","ref_index":47,"is_internal_anchor":false},{"citing_arxiv_id":"2510.15079","citing_title":"Assessing Coherency and Consistency of Code Execution Reasoning by Large Language Models","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2305.16264","citing_title":"Scaling Data-Constrained Language Models","ref_index":78,"is_internal_anchor":false},{"citing_arxiv_id":"2406.11794","citing_title":"DataComp-LM: In search of the next generation of training sets for language models","ref_index":129,"is_internal_anchor":false},{"citing_arxiv_id":"2602.01785","citing_title":"CodeOCR: On the Effectiveness of Vision Language Models in Code Understanding","ref_index":65,"is_internal_anchor":false},{"citing_arxiv_id":"2604.09557","citing_title":"SPEED-Bench: A Unified and Diverse Benchmark for Speculative Decoding","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2605.14350","citing_title":"Distributionally Robust Multi-Task Reinforcement Learning via Adaptive Task Sampling","ref_index":265,"is_internal_anchor":false},{"citing_arxiv_id":"2406.00515","citing_title":"A Survey on Large Language Models for Code Generation","ref_index":191,"is_internal_anchor":false},{"citing_arxiv_id":"2309.07597","citing_title":"C-Pack: Packed Resources For General Chinese Embeddings","ref_index":36,"is_internal_anchor":false},{"citing_arxiv_id":"2502.05171","citing_title":"Scaling up Test-Time Compute with Latent Reasoning: A Recurrent Depth Approach","ref_index":115,"is_internal_anchor":false},{"citing_arxiv_id":"2604.11770","citing_title":"Enhancing Program Repair with Specification Guidance and Intermediate Behavioral Signals","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07345","citing_title":"Mean-Pooled Cosine Similarity is Not Length-Invariant: Theory and Cross-Domain Evidence for a Length-Invariant Alternative","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2308.00352","citing_title":"MetaGPT: Meta Programming for A Multi-Agent Collaborative Framework","ref_index":38,"is_internal_anchor":false},{"citing_arxiv_id":"2305.06161","citing_title":"StarCoder: may the source be with you!","ref_index":297,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/OVAEHT7GC6T3QQHLEW5BFI3GRJ","json":"https://pith.science/pith/OVAEHT7GC6T3QQHLEW5BFI3GRJ.json","graph_json":"https://pith.science/api/pith-number/OVAEHT7GC6T3QQHLEW5BFI3GRJ/graph.json","events_json":"https://pith.science/api/pith-number/OVAEHT7GC6T3QQHLEW5BFI3GRJ/events.json","paper":"https://pith.science/paper/OVAEHT7G"},"agent_actions":{"view_html":"https://pith.science/pith/OVAEHT7GC6T3QQHLEW5BFI3GRJ","download_json":"https://pith.science/pith/OVAEHT7GC6T3QQHLEW5BFI3GRJ.json","view_paper":"https://pith.science/paper/OVAEHT7G","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2308.07124&json=true","fetch_graph":"https://pith.science/api/pith-number/OVAEHT7GC6T3QQHLEW5BFI3GRJ/graph.json","fetch_events":"https://pith.science/api/pith-number/OVAEHT7GC6T3QQHLEW5BFI3GRJ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/OVAEHT7GC6T3QQHLEW5BFI3GRJ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/OVAEHT7GC6T3QQHLEW5BFI3GRJ/action/storage_attestation","attest_author":"https://pith.science/pith/OVAEHT7GC6T3QQHLEW5BFI3GRJ/action/author_attestation","sign_citation":"https://pith.science/pith/OVAEHT7GC6T3QQHLEW5BFI3GRJ/action/citation_signature","submit_replication":"https://pith.science/pith/OVAEHT7GC6T3QQHLEW5BFI3GRJ/action/replication_record"}},"created_at":"2026-07-05T07:46:21.815169+00:00","updated_at":"2026-07-05T07:46:21.815169+00:00"}