{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2024:A3GL22GUHUUPGBKD3RJPDDSSP6","short_pith_number":"pith:A3GL22GU","canonical_record":{"source":{"id":"2401.02954","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-01-05T18:59:13Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"9651a793a7c3d28f987e13717456827b5d2fde86c16fa14df30f1eb8f64fd31e","abstract_canon_sha256":"49633e3c168d9b8e5e90202db26a797c9eec5a4c05fd2d80c21145feb52c02a5"},"schema_version":"1.0"},"canonical_sha256":"06ccbd68d43d28f30543dc52f18e527f898a51540fc3ff7a81e90a51990d3340","source":{"kind":"arxiv","id":"2401.02954","version":1},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2401.02954","created_at":"2026-07-05T07:30:40Z"},{"alias_kind":"arxiv_version","alias_value":"2401.02954v1","created_at":"2026-07-05T07:30:40Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2401.02954","created_at":"2026-07-05T07:30:40Z"},{"alias_kind":"pith_short_12","alias_value":"A3GL22GUHUUP","created_at":"2026-07-05T07:30:40Z"},{"alias_kind":"pith_short_16","alias_value":"A3GL22GUHUUPGBKD","created_at":"2026-07-05T07:30:40Z"},{"alias_kind":"pith_short_8","alias_value":"A3GL22GU","created_at":"2026-07-05T07:30:40Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2024:A3GL22GUHUUPGBKD3RJPDDSSP6","target":"record","payload":{"canonical_record":{"source":{"id":"2401.02954","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-01-05T18:59:13Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"9651a793a7c3d28f987e13717456827b5d2fde86c16fa14df30f1eb8f64fd31e","abstract_canon_sha256":"49633e3c168d9b8e5e90202db26a797c9eec5a4c05fd2d80c21145feb52c02a5"},"schema_version":"1.0"},"canonical_sha256":"06ccbd68d43d28f30543dc52f18e527f898a51540fc3ff7a81e90a51990d3340","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:30:40.867530Z","signature_b64":"kcGSgoYVigiIN67irvF5yP/vbeBm+Cu5wBQGoMksjFUIO5f2aDeFFWc6NXTxUJaISUpzPuclEJUt2b8+b63CBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"06ccbd68d43d28f30543dc52f18e527f898a51540fc3ff7a81e90a51990d3340","last_reissued_at":"2026-07-05T07:30:40.867082Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:30:40.867082Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2401.02954","source_version":1,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T07:30:40Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"z5mP8PvCpgk0ErvjALHZJh+XV28apnLVnVj5cp2G+B6JBVf42CLU72Ogxv9dzk3ZfRTXQ2ell6g9XcMwE82kCg==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-12T19:03:19.166933Z"},"content_sha256":"e11e0805343472d56c54837896ad1c7c3d9999acb028480ce1847880cf0fd4f8","schema_version":"1.0","event_id":"sha256:e11e0805343472d56c54837896ad1c7c3d9999acb028480ce1847880cf0fd4f8"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2024:A3GL22GUHUUPGBKD3RJPDDSSP6","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"DeepSeek LLM: Scaling Open-Source Language Models with Longtermism","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"DeepSeek LLM 67B surpasses LLaMA-2 70B on code, mathematics and reasoning benchmarks, with its chat version exceeding GPT-3.5 in open-ended evaluations.","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"A.X. Liu, Bingxuan Wang, Bo Liu, B. Zhang, Chenggang Zhao, Chengqi Deng, Chong Ruan, Damai Dai, Daya Guo, DeepSeek-AI: Xiao Bi, Dejian Yang, Deli Chen, Erhang Li, Fangyun Lin, Fuli Luo, Guangbo Hao, Guanting Chen, Guowei Li, Hanwei Xu, Haowei Zhang, Haoyu Lu, Honghui Ding, Huazuo Gao, Hui Qu, Jianzhong Guo, Jiashi Li, Jingxiang Sun, Junjie Qiu, Junxiao Song, Kai Dong, Kaige Gao, Kang Guan, Lecong Zhang, Liyue Zhang, Mingchuan Zhang, Minghua Zhang, Minghui Tang, Panpan Huang, Peiyi Wang, Qihao Zhu, Qiushi Du, Ruiqi Ge, R.X. Xu, Shanghao Lu, Shangyan Zhou, Shanhuang Chen, Shirong Ma, Shiyu Wang, Shuiping Yu, Shunfeng Zhou, Tian Pei, Tong Wu, Tongzheng Ren, Wenfeng Liang, Wenjie Hu, Wenjun Gao, Wen Liu, Wentao Zhang, XiaoDong Liu, Xiaotao Nie, Xingkai Yu, Xin Liu, Xin Xie, Xuecheng Su, Yanhong Xu, Yaofeng Sun, Yaohui Wang, Yao Li, Yao Zhao, Yichao Zhang, Yiliang Xiong, Ying He, Yishi Piao, Yiyuan Liu, Y.K. Li, Yongji Wang, Yuheng Zou, Yuxiang You, Y. Wu, Zehui Ren, Zhangli Sha, Zhe Fu, Zhenda Xie, Zhewen Hao, Zhihong Shao, Ziwei Xie","submitted_at":"2024-01-05T18:59:13Z","abstract_excerpt":"The rapid development of open-source large language models (LLMs) has been truly remarkable. However, the scaling law described in previous literature presents varying conclusions, which casts a dark cloud over scaling LLMs. We delve into the study of scaling laws and present our distinctive findings that facilitate scaling of large scale models in two commonly used open-source configurations, 7B and 67B. Guided by the scaling laws, we introduce DeepSeek LLM, a project dedicated to advancing open-source language models with a long-term perspective. To support the pre-training phase, we have de"},"claims":{"count":4,"items":[{"kind":"strongest_claim","text":"Our evaluation results demonstrate that DeepSeek LLM 67B surpasses LLaMA-2 70B on various benchmarks, particularly in the domains of code, mathematics, and reasoning. Furthermore, open-ended evaluations reveal that DeepSeek LLM 67B Chat exhibits superior performance compared to GPT-3.5.","source":"verdict.strongest_claim","status":"machine_extracted","claim_id":"C1","attestation":"unclaimed"},{"kind":"weakest_assumption","text":"The assumption that the chosen benchmarks and open-ended evaluations are fair, representative, and free of undisclosed advantages in training data, compute, or evaluation methodology.","source":"verdict.weakest_assumption","status":"machine_extracted","claim_id":"C2","attestation":"unclaimed"},{"kind":"one_line_summary","text":"DeepSeek LLM 67B exceeds LLaMA-2 70B on code, mathematics and reasoning benchmarks after pre-training on 2 trillion tokens and alignment via SFT and DPO.","source":"verdict.one_line_summary","status":"machine_extracted","claim_id":"C3","attestation":"unclaimed"},{"kind":"headline","text":"DeepSeek LLM 67B surpasses LLaMA-2 70B on code, mathematics and reasoning benchmarks, with its chat version exceeding GPT-3.5 in open-ended evaluations.","source":"verdict.pith_extraction.headline","status":"machine_extracted","claim_id":"C4","attestation":"unclaimed"}],"snapshot_sha256":"777c34686601c1f4da4d038aecd320c9a45245bc2b2bfdea681c2fefe5f39a44"},"source":{"id":"2401.02954","kind":"arxiv","version":1},"verdict":{"id":"6414bdc4-2d5d-4e45-9246-1040f10a9083","model_set":{"reader":"grok-4.3"},"created_at":"2026-05-11T06:03:01.583359Z","strongest_claim":"Our evaluation results demonstrate that DeepSeek LLM 67B surpasses LLaMA-2 70B on various benchmarks, particularly in the domains of code, mathematics, and reasoning. Furthermore, open-ended evaluations reveal that DeepSeek LLM 67B Chat exhibits superior performance compared to GPT-3.5.","one_line_summary":"DeepSeek LLM 67B exceeds LLaMA-2 70B on code, mathematics and reasoning benchmarks after pre-training on 2 trillion tokens and alignment via SFT and DPO.","pipeline_version":"pith-pipeline@v0.9.0","weakest_assumption":"The assumption that the chosen benchmarks and open-ended evaluations are fair, representative, and free of undisclosed advantages in training data, compute, or evaluation methodology.","pith_extraction_headline":"DeepSeek LLM 67B surpasses LLaMA-2 70B on code, mathematics and reasoning benchmarks, with its chat version exceeding GPT-3.5 in open-ended evaluations."},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2401.02954/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":128,"sample":[{"doi":"","year":2023,"title":"Introducing Claude","work_id":"2fb5bf54-38ba-478b-b07c-a6aa36421caf","ref_index":2,"cited_arxiv_id":"","is_internal_anchor":false},{"doi":"","year":2020,"title":"T. B. Brown, B. Mann, N. Ryder, M. Subbiah, J. Kaplan, P. Dhariwal, A. Neelakantan, P. Shyam, G. Sastry, A. Askell, S. Agarwal, A. Herbert-Voss, G. Krueger, T. Henighan, R. Child, A. Ramesh, D. M. Zie","work_id":"92c72853-9f8a-439c-a30c-9b82c858be39","ref_index":6,"cited_arxiv_id":"","is_internal_anchor":false},{"doi":"","year":2021,"title":"Evaluating Large Language Models Trained on Code","work_id":"042493e9-b26f-4b4e-bbde-382072ca9b08","ref_index":7,"cited_arxiv_id":"2107.03374","is_internal_anchor":true},{"doi":"","year":2023,"title":"T. Computer. Redpajama: an open dataset for training large language models, 2023. URL https://github.com/togethercomputer/RedPajama-Data","work_id":"325272d8-b15f-4a6e-a643-08d90fd4d30a","ref_index":10,"cited_arxiv_id":"","is_internal_anchor":false},{"doi":"","year":2023,"title":"T. Dao. Flash A ttention-2: Faster attention with better parallelism and work partitioning. 2023","work_id":"8f9fbc3f-d90a-4a61-b1b7-ddec23918f87","ref_index":12,"cited_arxiv_id":"","is_internal_anchor":false}],"resolved_work":128,"snapshot_sha256":"20bdd2fb975cec4f38cb88691224f28a0eb4f20eefbfab6e0df0c4333e078964","internal_anchors":37},"formal_canon":{"evidence_count":3,"snapshot_sha256":"3c30ef78fdc4064f1fc3303aaf8cc2f75cce321897961876bec08955ee0ff8f8"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":"6414bdc4-2d5d-4e45-9246-1040f10a9083"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T07:30:40Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"UJlGtlwZVEf2o/XCB5er42BDfmZHX/J1GHGTf8WuVu00toQfQ1kD7Uz8tNDMWOkhvlrnagrRJcS3eoef7qprCQ==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-12T19:03:19.168669Z"},"content_sha256":"f1509a9252ac5be13048bb8ed760b7878f00a06a732056b4143ca9d7120f8e0a","schema_version":"1.0","event_id":"sha256:f1509a9252ac5be13048bb8ed760b7878f00a06a732056b4143ca9d7120f8e0a"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/A3GL22GUHUUPGBKD3RJPDDSSP6/bundle.json","state_url":"https://pith.science/pith/A3GL22GUHUUPGBKD3RJPDDSSP6/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/A3GL22GUHUUPGBKD3RJPDDSSP6/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-12T19:03:19Z","links":{"resolver":"https://pith.science/pith/A3GL22GUHUUPGBKD3RJPDDSSP6","bundle":"https://pith.science/pith/A3GL22GUHUUPGBKD3RJPDDSSP6/bundle.json","state":"https://pith.science/pith/A3GL22GUHUUPGBKD3RJPDDSSP6/state.json","well_known_bundle":"https://pith.science/.well-known/pith/A3GL22GUHUUPGBKD3RJPDDSSP6/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2024:A3GL22GUHUUPGBKD3RJPDDSSP6","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"49633e3c168d9b8e5e90202db26a797c9eec5a4c05fd2d80c21145feb52c02a5","cross_cats_sorted":["cs.AI","cs.LG"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-01-05T18:59:13Z","title_canon_sha256":"9651a793a7c3d28f987e13717456827b5d2fde86c16fa14df30f1eb8f64fd31e"},"schema_version":"1.0","source":{"id":"2401.02954","kind":"arxiv","version":1}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2401.02954","created_at":"2026-07-05T07:30:40Z"},{"alias_kind":"arxiv_version","alias_value":"2401.02954v1","created_at":"2026-07-05T07:30:40Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2401.02954","created_at":"2026-07-05T07:30:40Z"},{"alias_kind":"pith_short_12","alias_value":"A3GL22GUHUUP","created_at":"2026-07-05T07:30:40Z"},{"alias_kind":"pith_short_16","alias_value":"A3GL22GUHUUPGBKD","created_at":"2026-07-05T07:30:40Z"},{"alias_kind":"pith_short_8","alias_value":"A3GL22GU","created_at":"2026-07-05T07:30:40Z"}],"graph_snapshots":[{"event_id":"sha256:f1509a9252ac5be13048bb8ed760b7878f00a06a732056b4143ca9d7120f8e0a","target":"graph","created_at":"2026-07-05T07:30:40Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":4,"items":[{"attestation":"unclaimed","claim_id":"C1","kind":"strongest_claim","source":"verdict.strongest_claim","status":"machine_extracted","text":"Our evaluation results demonstrate that DeepSeek LLM 67B surpasses LLaMA-2 70B on various benchmarks, particularly in the domains of code, mathematics, and reasoning. Furthermore, open-ended evaluations reveal that DeepSeek LLM 67B Chat exhibits superior performance compared to GPT-3.5."},{"attestation":"unclaimed","claim_id":"C2","kind":"weakest_assumption","source":"verdict.weakest_assumption","status":"machine_extracted","text":"The assumption that the chosen benchmarks and open-ended evaluations are fair, representative, and free of undisclosed advantages in training data, compute, or evaluation methodology."},{"attestation":"unclaimed","claim_id":"C3","kind":"one_line_summary","source":"verdict.one_line_summary","status":"machine_extracted","text":"DeepSeek LLM 67B exceeds LLaMA-2 70B on code, mathematics and reasoning benchmarks after pre-training on 2 trillion tokens and alignment via SFT and DPO."},{"attestation":"unclaimed","claim_id":"C4","kind":"headline","source":"verdict.pith_extraction.headline","status":"machine_extracted","text":"DeepSeek LLM 67B surpasses LLaMA-2 70B on code, mathematics and reasoning benchmarks, with its chat version exceeding GPT-3.5 in open-ended evaluations."}],"snapshot_sha256":"777c34686601c1f4da4d038aecd320c9a45245bc2b2bfdea681c2fefe5f39a44"},"formal_canon":{"evidence_count":3,"snapshot_sha256":"3c30ef78fdc4064f1fc3303aaf8cc2f75cce321897961876bec08955ee0ff8f8"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2401.02954/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"The rapid development of open-source large language models (LLMs) has been truly remarkable. However, the scaling law described in previous literature presents varying conclusions, which casts a dark cloud over scaling LLMs. We delve into the study of scaling laws and present our distinctive findings that facilitate scaling of large scale models in two commonly used open-source configurations, 7B and 67B. Guided by the scaling laws, we introduce DeepSeek LLM, a project dedicated to advancing open-source language models with a long-term perspective. To support the pre-training phase, we have de","authors_text":"A.X. Liu, Bingxuan Wang, Bo Liu, B. Zhang, Chenggang Zhao, Chengqi Deng, Chong Ruan, Damai Dai, Daya Guo, DeepSeek-AI: Xiao Bi, Dejian Yang, Deli Chen, Erhang Li, Fangyun Lin, Fuli Luo, Guangbo Hao, Guanting Chen, Guowei Li, Hanwei Xu, Haowei Zhang, Haoyu Lu, Honghui Ding, Huazuo Gao, Hui Qu, Jianzhong Guo, Jiashi Li, Jingxiang Sun, Junjie Qiu, Junxiao Song, Kai Dong, Kaige Gao, Kang Guan, Lecong Zhang, Liyue Zhang, Mingchuan Zhang, Minghua Zhang, Minghui Tang, Panpan Huang, Peiyi Wang, Qihao Zhu, Qiushi Du, Ruiqi Ge, R.X. Xu, Shanghao Lu, Shangyan Zhou, Shanhuang Chen, Shirong Ma, Shiyu Wang, Shuiping Yu, Shunfeng Zhou, Tian Pei, Tong Wu, Tongzheng Ren, Wenfeng Liang, Wenjie Hu, Wenjun Gao, Wen Liu, Wentao Zhang, XiaoDong Liu, Xiaotao Nie, Xingkai Yu, Xin Liu, Xin Xie, Xuecheng Su, Yanhong Xu, Yaofeng Sun, Yaohui Wang, Yao Li, Yao Zhao, Yichao Zhang, Yiliang Xiong, Ying He, Yishi Piao, Yiyuan Liu, Y.K. Li, Yongji Wang, Yuheng Zou, Yuxiang You, Y. Wu, Zehui Ren, Zhangli Sha, Zhe Fu, Zhenda Xie, Zhewen Hao, Zhihong Shao, Ziwei Xie","cross_cats":["cs.AI","cs.LG"],"headline":"DeepSeek LLM 67B surpasses LLaMA-2 70B on code, mathematics and reasoning benchmarks, with its chat version exceeding GPT-3.5 in open-ended evaluations.","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-01-05T18:59:13Z","title":"DeepSeek LLM: Scaling Open-Source Language Models with Longtermism"},"references":{"count":128,"internal_anchors":37,"resolved_work":128,"sample":[{"cited_arxiv_id":"","doi":"","is_internal_anchor":false,"ref_index":2,"title":"Introducing Claude","work_id":"2fb5bf54-38ba-478b-b07c-a6aa36421caf","year":2023},{"cited_arxiv_id":"","doi":"","is_internal_anchor":false,"ref_index":6,"title":"T. B. Brown, B. Mann, N. Ryder, M. Subbiah, J. Kaplan, P. Dhariwal, A. Neelakantan, P. Shyam, G. Sastry, A. Askell, S. Agarwal, A. Herbert-Voss, G. Krueger, T. Henighan, R. Child, A. Ramesh, D. M. Zie","work_id":"92c72853-9f8a-439c-a30c-9b82c858be39","year":2020},{"cited_arxiv_id":"2107.03374","doi":"","is_internal_anchor":true,"ref_index":7,"title":"Evaluating Large Language Models Trained on Code","work_id":"042493e9-b26f-4b4e-bbde-382072ca9b08","year":2021},{"cited_arxiv_id":"","doi":"","is_internal_anchor":false,"ref_index":10,"title":"T. Computer. Redpajama: an open dataset for training large language models, 2023. URL https://github.com/togethercomputer/RedPajama-Data","work_id":"325272d8-b15f-4a6e-a643-08d90fd4d30a","year":2023},{"cited_arxiv_id":"","doi":"","is_internal_anchor":false,"ref_index":12,"title":"T. Dao. Flash A ttention-2: Faster attention with better parallelism and work partitioning. 2023","work_id":"8f9fbc3f-d90a-4a61-b1b7-ddec23918f87","year":2023}],"snapshot_sha256":"20bdd2fb975cec4f38cb88691224f28a0eb4f20eefbfab6e0df0c4333e078964"},"source":{"id":"2401.02954","kind":"arxiv","version":1},"verdict":{"created_at":"2026-05-11T06:03:01.583359Z","id":"6414bdc4-2d5d-4e45-9246-1040f10a9083","model_set":{"reader":"grok-4.3"},"one_line_summary":"DeepSeek LLM 67B exceeds LLaMA-2 70B on code, mathematics and reasoning benchmarks after pre-training on 2 trillion tokens and alignment via SFT and DPO.","pipeline_version":"pith-pipeline@v0.9.0","pith_extraction_headline":"DeepSeek LLM 67B surpasses LLaMA-2 70B on code, mathematics and reasoning benchmarks, with its chat version exceeding GPT-3.5 in open-ended evaluations.","strongest_claim":"Our evaluation results demonstrate that DeepSeek LLM 67B surpasses LLaMA-2 70B on various benchmarks, particularly in the domains of code, mathematics, and reasoning. Furthermore, open-ended evaluations reveal that DeepSeek LLM 67B Chat exhibits superior performance compared to GPT-3.5.","weakest_assumption":"The assumption that the chosen benchmarks and open-ended evaluations are fair, representative, and free of undisclosed advantages in training data, compute, or evaluation methodology."}},"verdict_id":"6414bdc4-2d5d-4e45-9246-1040f10a9083"}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:e11e0805343472d56c54837896ad1c7c3d9999acb028480ce1847880cf0fd4f8","target":"record","created_at":"2026-07-05T07:30:40Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"49633e3c168d9b8e5e90202db26a797c9eec5a4c05fd2d80c21145feb52c02a5","cross_cats_sorted":["cs.AI","cs.LG"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-01-05T18:59:13Z","title_canon_sha256":"9651a793a7c3d28f987e13717456827b5d2fde86c16fa14df30f1eb8f64fd31e"},"schema_version":"1.0","source":{"id":"2401.02954","kind":"arxiv","version":1}},"canonical_sha256":"06ccbd68d43d28f30543dc52f18e527f898a51540fc3ff7a81e90a51990d3340","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"06ccbd68d43d28f30543dc52f18e527f898a51540fc3ff7a81e90a51990d3340","first_computed_at":"2026-07-05T07:30:40.867082Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T07:30:40.867082Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"kcGSgoYVigiIN67irvF5yP/vbeBm+Cu5wBQGoMksjFUIO5f2aDeFFWc6NXTxUJaISUpzPuclEJUt2b8+b63CBg==","signature_status":"signed_v1","signed_at":"2026-07-05T07:30:40.867530Z","signed_message":"canonical_sha256_bytes"},"source_id":"2401.02954","source_kind":"arxiv","source_version":1}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:e11e0805343472d56c54837896ad1c7c3d9999acb028480ce1847880cf0fd4f8","sha256:f1509a9252ac5be13048bb8ed760b7878f00a06a732056b4143ca9d7120f8e0a"],"state_sha256":"455383b72e12e3d8c65121221ebee8ae80b811a764c49684f578964c137005c9"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"RNHNoV8qtmuA+NBVlT69Sk886jAKXcTPodAZv7GEB1+kqtDBbtvGVoPHdUirTrUSq1ob1tgcdrbTagxDqtTlDA==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-12T19:03:19.231140Z","bundle_sha256":"f43b85f39d0ea2ebac7f3f4680a4dd4de238f13cabbe6eb300c7b9988bc1044e"}}