{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:SM2WWDC5XDBNQ2ZOEUQRPBUZJP","short_pith_number":"pith:SM2WWDC5","schema_version":"1.0","canonical_sha256":"93356b0c5db8c2d86b2e25211786994be43919395193de0e52bf0aadd77ae31d","source":{"kind":"arxiv","id":"2412.17395","version":3},"attestation_state":"computed","paper":{"title":"WarriorCoder: Learning from Expert Battles to Augment Code Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Can Xu, Dongmei Zhang, Fangkai Yang, Huawen Feng, Lu Wang, Pu Zhao, Qianli Ma, Qingfeng Sun, Qingwei Lin, Qi Zhang, Saravan Rajmohan","submitted_at":"2024-12-23T08:47:42Z","abstract_excerpt":"Despite recent progress achieved by code large language models (LLMs), their remarkable abilities are largely dependent on fine-tuning on the high-quality data, posing challenges for data collection and annotation. To address this, current methods often design various data flywheels to collect complex code instructions, enabling models to handle more intricate tasks. However, these approaches typically rely on off-the-shelf datasets and data augmentation from a limited set of proprietary LLMs (e.g., Claude, GPT4, and so on), which restricts the diversity of the constructed data and makes it pr"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2412.17395","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-12-23T08:47:42Z","cross_cats_sorted":[],"title_canon_sha256":"e7ad4ad3e2cf8d7e2146560e19dceecdee7c940594107bda7bc7c62e35c2391f","abstract_canon_sha256":"20108d9539e0ae7446861cae7530ee2b007e8b1736979e13687d6329dd47d64d"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:15:48.146369Z","signature_b64":"lEOsQpaQML6CXSOTEIAHDXDLG1siZEp2wvnEToNGiUlVRxz4kyUGIa+3YI/2JmR9naiR0bbHs5OtwF937wsoCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"93356b0c5db8c2d86b2e25211786994be43919395193de0e52bf0aadd77ae31d","last_reissued_at":"2026-07-05T10:15:48.145708Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:15:48.145708Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"WarriorCoder: Learning from Expert Battles to Augment Code Large Language Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Can Xu, Dongmei Zhang, Fangkai Yang, Huawen Feng, Lu Wang, Pu Zhao, Qianli Ma, Qingfeng Sun, Qingwei Lin, Qi Zhang, Saravan Rajmohan","submitted_at":"2024-12-23T08:47:42Z","abstract_excerpt":"Despite recent progress achieved by code large language models (LLMs), their remarkable abilities are largely dependent on fine-tuning on the high-quality data, posing challenges for data collection and annotation. To address this, current methods often design various data flywheels to collect complex code instructions, enabling models to handle more intricate tasks. However, these approaches typically rely on off-the-shelf datasets and data augmentation from a limited set of proprietary LLMs (e.g., Claude, GPT4, and so on), which restricts the diversity of the constructed data and makes it pr"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2412.17395","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2412.17395/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2412.17395","created_at":"2026-07-05T10:15:48.145852+00:00"},{"alias_kind":"arxiv_version","alias_value":"2412.17395v3","created_at":"2026-07-05T10:15:48.145852+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2412.17395","created_at":"2026-07-05T10:15:48.145852+00:00"},{"alias_kind":"pith_short_12","alias_value":"SM2WWDC5XDBN","created_at":"2026-07-05T10:15:48.145852+00:00"},{"alias_kind":"pith_short_16","alias_value":"SM2WWDC5XDBNQ2ZO","created_at":"2026-07-05T10:15:48.145852+00:00"},{"alias_kind":"pith_short_8","alias_value":"SM2WWDC5","created_at":"2026-07-05T10:15:48.145852+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/SM2WWDC5XDBNQ2ZOEUQRPBUZJP","json":"https://pith.science/pith/SM2WWDC5XDBNQ2ZOEUQRPBUZJP.json","graph_json":"https://pith.science/api/pith-number/SM2WWDC5XDBNQ2ZOEUQRPBUZJP/graph.json","events_json":"https://pith.science/api/pith-number/SM2WWDC5XDBNQ2ZOEUQRPBUZJP/events.json","paper":"https://pith.science/paper/SM2WWDC5"},"agent_actions":{"view_html":"https://pith.science/pith/SM2WWDC5XDBNQ2ZOEUQRPBUZJP","download_json":"https://pith.science/pith/SM2WWDC5XDBNQ2ZOEUQRPBUZJP.json","view_paper":"https://pith.science/paper/SM2WWDC5","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2412.17395&json=true","fetch_graph":"https://pith.science/api/pith-number/SM2WWDC5XDBNQ2ZOEUQRPBUZJP/graph.json","fetch_events":"https://pith.science/api/pith-number/SM2WWDC5XDBNQ2ZOEUQRPBUZJP/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/SM2WWDC5XDBNQ2ZOEUQRPBUZJP/action/timestamp_anchor","attest_storage":"https://pith.science/pith/SM2WWDC5XDBNQ2ZOEUQRPBUZJP/action/storage_attestation","attest_author":"https://pith.science/pith/SM2WWDC5XDBNQ2ZOEUQRPBUZJP/action/author_attestation","sign_citation":"https://pith.science/pith/SM2WWDC5XDBNQ2ZOEUQRPBUZJP/action/citation_signature","submit_replication":"https://pith.science/pith/SM2WWDC5XDBNQ2ZOEUQRPBUZJP/action/replication_record"}},"created_at":"2026-07-05T10:15:48.145852+00:00","updated_at":"2026-07-05T10:15:48.145852+00:00"}