{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2024:A2FM4SHKWWDIK3MMFY3NPI5NBR","short_pith_number":"pith:A2FM4SHK","canonical_record":{"source":{"id":"2405.04434","kind":"arxiv","version":5},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-05-07T15:56:43Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"a4fbad70e2dec8122e1eb20bbd56a6f5df3fa6e0323eb665c0516c199914368f","abstract_canon_sha256":"aa1f21fe687cbf7b5fce7a5b9ce802a45332d6122c30914c866eaf54ca837e83"},"schema_version":"1.0"},"canonical_sha256":"068ace48eab586856d8c2e36d7a3ad0c5ec333f6210d2271dd582973259b8c11","source":{"kind":"arxiv","id":"2405.04434","version":5},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2405.04434","created_at":"2026-07-05T08:34:18Z"},{"alias_kind":"arxiv_version","alias_value":"2405.04434v5","created_at":"2026-07-05T08:34:18Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2405.04434","created_at":"2026-07-05T08:34:18Z"},{"alias_kind":"pith_short_12","alias_value":"A2FM4SHKWWDI","created_at":"2026-07-05T08:34:18Z"},{"alias_kind":"pith_short_16","alias_value":"A2FM4SHKWWDIK3MM","created_at":"2026-07-05T08:34:18Z"},{"alias_kind":"pith_short_8","alias_value":"A2FM4SHK","created_at":"2026-07-05T08:34:18Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2024:A2FM4SHKWWDIK3MMFY3NPI5NBR","target":"record","payload":{"canonical_record":{"source":{"id":"2405.04434","kind":"arxiv","version":5},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-05-07T15:56:43Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"a4fbad70e2dec8122e1eb20bbd56a6f5df3fa6e0323eb665c0516c199914368f","abstract_canon_sha256":"aa1f21fe687cbf7b5fce7a5b9ce802a45332d6122c30914c866eaf54ca837e83"},"schema_version":"1.0"},"canonical_sha256":"068ace48eab586856d8c2e36d7a3ad0c5ec333f6210d2271dd582973259b8c11","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:34:18.327687Z","signature_b64":"tkjxqq3atgxP8LKobD54mZNVX2roNwVZTYH3cb8WSrWbKRZVc8CDlYeqgSdue5wWRfGJ6VBh+vaBveOW34pSCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"068ace48eab586856d8c2e36d7a3ad0c5ec333f6210d2271dd582973259b8c11","last_reissued_at":"2026-07-05T08:34:18.327108Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:34:18.327108Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2405.04434","source_version":5,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T08:34:18Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"18H5+LM5jopJgVwlAvvQFsVuAuUMMh3YvgQNySjQ3HzwgdIqEJPMs+egGck1z+i4xe5kJ7pW9HP44Pg10byBDQ==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-12T16:45:04.406529Z"},"content_sha256":"08486a4f15e9e7c9f5e7d6e929fa0a4b6736dffbf9d7de542f78232f789457ae","schema_version":"1.0","event_id":"sha256:08486a4f15e9e7c9f5e7d6e929fa0a4b6736dffbf9d7de542f78232f789457ae"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2024:A2FM4SHKWWDIK3MMFY3NPI5NBR","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"DeepSeek-V2: A Strong, Economical, and Efficient Mixture-of-Experts Language Model","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"DeepSeek-V2 shows a Mixture-of-Experts model with 236 billion total parameters but only 21 billion activated per token can match top open-source language models while lowering training costs and inference demands.","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Aixin Liu, Bei Feng, Bingxuan Wang, Bin Wang, Bo Liu, Chenggang Zhao, Chengqi Dengr, Chong Ruan, Damai Dai, Daya Guo, DeepSeek-AI, Dejian Yang, Deli Chen, Dongjie Ji, Erhang Li, Fangyun Lin, Fuli Luo, Guangbo Hao, Guanting Chen, Guowei Li, Hanwei Xu, Haowei Zhang, Hao Yang, Honghui Ding, Huajian Xin, Huazuo Gao, Hui Li, Hui Qu, H. Zhang, Jian Liang, Jianzhong Guo, Jiaqi Ni, Jiashi Li, Jin Chen, Jingyang Yuan, J.L. Cai, Junjie Qiu, Junxiao Song, Kai Dong, Kaige Gao, Kang Guan, Lean Wang, Lecong Zhang, Lei Xu, Leyi Xia, Liang Zhao, Liyue Zhang, Meng Li, Miaojun Wang, Mingchuan Zhang, Minghua Zhang, Minghui Tang, Mingming Li, Ning Tian, Panpan Huang, Peiyi Wang, Peng Zhang, Qihao Zhu, Qinyu Chen, Qiushi Du, R.J. Chen, R.L. Jin, Ruiqi Ge, Ruizhe Pan, Runxin Xu, Ruyi Chen, Shanghao Lu, Shangyan Zhou, Shanhuang Chen, Shaoqing Wu, Shengfeng Ye, Shirong Ma, Shiyu Wang, Shuang Zhou, Shuiping Yu, Shunfeng Zhou, Size Zheng, S.S. Li, Tian Pei, Tian Yuan, Tianyu Sun, T. Wang, Wangding Zeng, Wei An, Wenfeng Liang, Wenjun Gao, Wen Liu, Wentao Zhang, W.L. Xiao, Xiangyue Jin, Xianzu Wang, Xiao Bi, XiaoDong Liu, Xiaohan Wang, Xiaojin Shen, Xiaokang Chen, Xiaosha Chen, Xiaotao Nie, Xiaowen Sun, Xiaoxiang Wang, Xingkai Yu, Xin Liu, Xinnan Song, Xin Xie, Xinyi Zhou, Xinyu Yang, X.Q. Li, Xuan Lu, Xuecheng Su, Yanhong Xu, Yanping Huang, Yaofeng Sun, Yaohui Li, Yaohui Wang, Yao Li, Yao Zhao, Yichao Zhang, Yiliang Xiong, Yilong Zhao, Ying He, Ying Tang, Yishi Piao, Yixin Dong, Yixuan Tan, Yiyuan Liu, Yi Zheng, Y.K. Li, Yongji Wang, Yongqiang Guo, Yuchen Zhu, Yuduan Wang, Yuheng Zou, Yukun Zha, Yunxian Ma, Yuting Yan, Yuxiang You, Yuxuan Liu, Y. Wu, Y.X. Wei, Y.X. Zhu, Zehui Ren, Zhangli Sha, Zhe Fu, Zhenda Xie, Zhen Huang, Zhen Zhang, Zhewen Hao, Zhihong Shao, Zhiniu Wen, Zhipeng Xu, Zhongyu Zhang, Zhuoshu Li, Zihan Wang, Zihui Gu, Zilin Li, Ziwei Xie, Z.Z. Ren","submitted_at":"2024-05-07T15:56:43Z","abstract_excerpt":"We present DeepSeek-V2, a strong Mixture-of-Experts (MoE) language model characterized by economical training and efficient inference. It comprises 236B total parameters, of which 21B are activated for each token, and supports a context length of 128K tokens. DeepSeek-V2 adopts innovative architectures including Multi-head Latent Attention (MLA) and DeepSeekMoE. MLA guarantees efficient inference through significantly compressing the Key-Value (KV) cache into a latent vector, while DeepSeekMoE enables training strong models at an economical cost through sparse computation. Compared with DeepSe"},"claims":{"count":4,"items":[{"kind":"strongest_claim","text":"even with only 21B activated parameters, DeepSeek-V2 and its chat versions still achieve top-tier performance among open-source models.","source":"verdict.strongest_claim","status":"machine_extracted","claim_id":"C1","attestation":"unclaimed"},{"kind":"weakest_assumption","text":"That the reported performance and efficiency gains were measured against fair, standardized baselines without post-hoc data selection or undisclosed implementation advantages.","source":"verdict.weakest_assumption","status":"machine_extracted","claim_id":"C2","attestation":"unclaimed"},{"kind":"one_line_summary","text":"DeepSeek-V2 delivers top-tier open-source LLM performance using only 21B active parameters by compressing the KV cache 93.3% and cutting training costs 42.5% via MLA and DeepSeekMoE.","source":"verdict.one_line_summary","status":"machine_extracted","claim_id":"C3","attestation":"unclaimed"},{"kind":"headline","text":"DeepSeek-V2 shows a Mixture-of-Experts model with 236 billion total parameters but only 21 billion activated per token can match top open-source language models while lowering training costs and inference demands.","source":"verdict.pith_extraction.headline","status":"machine_extracted","claim_id":"C4","attestation":"unclaimed"}],"snapshot_sha256":"d97c0d0a5eb917a824b78596538d02c99695da0b42a50d6be16b3ecb259d4ff5"},"source":{"id":"2405.04434","kind":"arxiv","version":5},"verdict":{"id":"6bb7c623-af08-4124-8aec-ab0989e4dd26","model_set":{"reader":"grok-4.3"},"created_at":"2026-05-11T05:30:50.710829Z","strongest_claim":"even with only 21B activated parameters, DeepSeek-V2 and its chat versions still achieve top-tier performance among open-source models.","one_line_summary":"DeepSeek-V2 delivers top-tier open-source LLM performance using only 21B active parameters by compressing the KV cache 93.3% and cutting training costs 42.5% via MLA and DeepSeekMoE.","pipeline_version":"pith-pipeline@v0.9.0","weakest_assumption":"That the reported performance and efficiency gains were measured against fair, standardized baselines without post-hoc data selection or undisclosed implementation advantages.","pith_extraction_headline":"DeepSeek-V2 shows a Mixture-of-Experts model with 236 billion total parameters but only 21 billion activated per token can match top open-source language models while lowering training costs and inference demands."},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2405.04434/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":150,"sample":[{"doi":"","year":2024,"title":"Llama 3 model card","work_id":"008d23c2-07d0-4784-a704-56521d627c6b","ref_index":1,"cited_arxiv_id":"","is_internal_anchor":false},{"doi":"","year":2023,"title":"Introducing Claude","work_id":"2fb5bf54-38ba-478b-b07c-a6aa36421caf","ref_index":3,"cited_arxiv_id":"","is_internal_anchor":false},{"doi":"","year":2021,"title":"Evaluating Large Language Models Trained on Code","work_id":"042493e9-b26f-4b4e-bbde-382072ca9b08","ref_index":7,"cited_arxiv_id":"2107.03374","is_internal_anchor":true},{"doi":"10.48550/arxiv.2401.06066","year":2024,"title":"DeepSeekMoE: Towards Ultimate Expert Specialization in Mixture-of-Experts Language Models","work_id":"a9888d6d-bf47-4324-9834-7cc12ac3a78c","ref_index":11,"cited_arxiv_id":"2401.06066","is_internal_anchor":true},{"doi":"","year":2023,"title":"T. Dao. Flash A ttention-2: Faster attention with better parallelism and work partitioning, 2023","work_id":"571162b3-cda9-4125-b3d2-5904014985c1","ref_index":12,"cited_arxiv_id":"","is_internal_anchor":false}],"resolved_work":150,"snapshot_sha256":"7439d9d44296192850ae12f7699688a3c6cf2264fe0baa11a449a5910375d04c","internal_anchors":48},"formal_canon":{"evidence_count":2,"snapshot_sha256":"942221f07dc6d872da979705c78ab6f7de146775cf7eda23030ec6caba6f8468"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":"6bb7c623-af08-4124-8aec-ab0989e4dd26"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T08:34:18Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"FKoy+JmmorKbYc3X/PYzBHqg0MUebYNVPzFWvJPMoljXY13uDCxpwDDzkYtTBg/93CK73913fDk1livpC76LCg==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-12T16:45:04.408059Z"},"content_sha256":"de1e1b2cb874a02f5107bca90d03a446ceb72405be7ef49959a7702f0a272fbc","schema_version":"1.0","event_id":"sha256:de1e1b2cb874a02f5107bca90d03a446ceb72405be7ef49959a7702f0a272fbc"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/A2FM4SHKWWDIK3MMFY3NPI5NBR/bundle.json","state_url":"https://pith.science/pith/A2FM4SHKWWDIK3MMFY3NPI5NBR/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/A2FM4SHKWWDIK3MMFY3NPI5NBR/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-12T16:45:04Z","links":{"resolver":"https://pith.science/pith/A2FM4SHKWWDIK3MMFY3NPI5NBR","bundle":"https://pith.science/pith/A2FM4SHKWWDIK3MMFY3NPI5NBR/bundle.json","state":"https://pith.science/pith/A2FM4SHKWWDIK3MMFY3NPI5NBR/state.json","well_known_bundle":"https://pith.science/.well-known/pith/A2FM4SHKWWDIK3MMFY3NPI5NBR/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2024:A2FM4SHKWWDIK3MMFY3NPI5NBR","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"aa1f21fe687cbf7b5fce7a5b9ce802a45332d6122c30914c866eaf54ca837e83","cross_cats_sorted":["cs.AI"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-05-07T15:56:43Z","title_canon_sha256":"a4fbad70e2dec8122e1eb20bbd56a6f5df3fa6e0323eb665c0516c199914368f"},"schema_version":"1.0","source":{"id":"2405.04434","kind":"arxiv","version":5}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2405.04434","created_at":"2026-07-05T08:34:18Z"},{"alias_kind":"arxiv_version","alias_value":"2405.04434v5","created_at":"2026-07-05T08:34:18Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2405.04434","created_at":"2026-07-05T08:34:18Z"},{"alias_kind":"pith_short_12","alias_value":"A2FM4SHKWWDI","created_at":"2026-07-05T08:34:18Z"},{"alias_kind":"pith_short_16","alias_value":"A2FM4SHKWWDIK3MM","created_at":"2026-07-05T08:34:18Z"},{"alias_kind":"pith_short_8","alias_value":"A2FM4SHK","created_at":"2026-07-05T08:34:18Z"}],"graph_snapshots":[{"event_id":"sha256:de1e1b2cb874a02f5107bca90d03a446ceb72405be7ef49959a7702f0a272fbc","target":"graph","created_at":"2026-07-05T08:34:18Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":4,"items":[{"attestation":"unclaimed","claim_id":"C1","kind":"strongest_claim","source":"verdict.strongest_claim","status":"machine_extracted","text":"even with only 21B activated parameters, DeepSeek-V2 and its chat versions still achieve top-tier performance among open-source models."},{"attestation":"unclaimed","claim_id":"C2","kind":"weakest_assumption","source":"verdict.weakest_assumption","status":"machine_extracted","text":"That the reported performance and efficiency gains were measured against fair, standardized baselines without post-hoc data selection or undisclosed implementation advantages."},{"attestation":"unclaimed","claim_id":"C3","kind":"one_line_summary","source":"verdict.one_line_summary","status":"machine_extracted","text":"DeepSeek-V2 delivers top-tier open-source LLM performance using only 21B active parameters by compressing the KV cache 93.3% and cutting training costs 42.5% via MLA and DeepSeekMoE."},{"attestation":"unclaimed","claim_id":"C4","kind":"headline","source":"verdict.pith_extraction.headline","status":"machine_extracted","text":"DeepSeek-V2 shows a Mixture-of-Experts model with 236 billion total parameters but only 21 billion activated per token can match top open-source language models while lowering training costs and inference demands."}],"snapshot_sha256":"d97c0d0a5eb917a824b78596538d02c99695da0b42a50d6be16b3ecb259d4ff5"},"formal_canon":{"evidence_count":2,"snapshot_sha256":"942221f07dc6d872da979705c78ab6f7de146775cf7eda23030ec6caba6f8468"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2405.04434/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"We present DeepSeek-V2, a strong Mixture-of-Experts (MoE) language model characterized by economical training and efficient inference. It comprises 236B total parameters, of which 21B are activated for each token, and supports a context length of 128K tokens. DeepSeek-V2 adopts innovative architectures including Multi-head Latent Attention (MLA) and DeepSeekMoE. MLA guarantees efficient inference through significantly compressing the Key-Value (KV) cache into a latent vector, while DeepSeekMoE enables training strong models at an economical cost through sparse computation. Compared with DeepSe","authors_text":"Aixin Liu, Bei Feng, Bingxuan Wang, Bin Wang, Bo Liu, Chenggang Zhao, Chengqi Dengr, Chong Ruan, Damai Dai, Daya Guo, DeepSeek-AI, Dejian Yang, Deli Chen, Dongjie Ji, Erhang Li, Fangyun Lin, Fuli Luo, Guangbo Hao, Guanting Chen, Guowei Li, Hanwei Xu, Haowei Zhang, Hao Yang, Honghui Ding, Huajian Xin, Huazuo Gao, Hui Li, Hui Qu, H. Zhang, Jian Liang, Jianzhong Guo, Jiaqi Ni, Jiashi Li, Jin Chen, Jingyang Yuan, J.L. Cai, Junjie Qiu, Junxiao Song, Kai Dong, Kaige Gao, Kang Guan, Lean Wang, Lecong Zhang, Lei Xu, Leyi Xia, Liang Zhao, Liyue Zhang, Meng Li, Miaojun Wang, Mingchuan Zhang, Minghua Zhang, Minghui Tang, Mingming Li, Ning Tian, Panpan Huang, Peiyi Wang, Peng Zhang, Qihao Zhu, Qinyu Chen, Qiushi Du, R.J. Chen, R.L. Jin, Ruiqi Ge, Ruizhe Pan, Runxin Xu, Ruyi Chen, Shanghao Lu, Shangyan Zhou, Shanhuang Chen, Shaoqing Wu, Shengfeng Ye, Shirong Ma, Shiyu Wang, Shuang Zhou, Shuiping Yu, Shunfeng Zhou, Size Zheng, S.S. Li, Tian Pei, Tian Yuan, Tianyu Sun, T. Wang, Wangding Zeng, Wei An, Wenfeng Liang, Wenjun Gao, Wen Liu, Wentao Zhang, W.L. Xiao, Xiangyue Jin, Xianzu Wang, Xiao Bi, XiaoDong Liu, Xiaohan Wang, Xiaojin Shen, Xiaokang Chen, Xiaosha Chen, Xiaotao Nie, Xiaowen Sun, Xiaoxiang Wang, Xingkai Yu, Xin Liu, Xinnan Song, Xin Xie, Xinyi Zhou, Xinyu Yang, X.Q. Li, Xuan Lu, Xuecheng Su, Yanhong Xu, Yanping Huang, Yaofeng Sun, Yaohui Li, Yaohui Wang, Yao Li, Yao Zhao, Yichao Zhang, Yiliang Xiong, Yilong Zhao, Ying He, Ying Tang, Yishi Piao, Yixin Dong, Yixuan Tan, Yiyuan Liu, Yi Zheng, Y.K. Li, Yongji Wang, Yongqiang Guo, Yuchen Zhu, Yuduan Wang, Yuheng Zou, Yukun Zha, Yunxian Ma, Yuting Yan, Yuxiang You, Yuxuan Liu, Y. Wu, Y.X. Wei, Y.X. Zhu, Zehui Ren, Zhangli Sha, Zhe Fu, Zhenda Xie, Zhen Huang, Zhen Zhang, Zhewen Hao, Zhihong Shao, Zhiniu Wen, Zhipeng Xu, Zhongyu Zhang, Zhuoshu Li, Zihan Wang, Zihui Gu, Zilin Li, Ziwei Xie, Z.Z. Ren","cross_cats":["cs.AI"],"headline":"DeepSeek-V2 shows a Mixture-of-Experts model with 236 billion total parameters but only 21 billion activated per token can match top open-source language models while lowering training costs and inference demands.","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-05-07T15:56:43Z","title":"DeepSeek-V2: A Strong, Economical, and Efficient Mixture-of-Experts Language Model"},"references":{"count":150,"internal_anchors":48,"resolved_work":150,"sample":[{"cited_arxiv_id":"","doi":"","is_internal_anchor":false,"ref_index":1,"title":"Llama 3 model card","work_id":"008d23c2-07d0-4784-a704-56521d627c6b","year":2024},{"cited_arxiv_id":"","doi":"","is_internal_anchor":false,"ref_index":3,"title":"Introducing Claude","work_id":"2fb5bf54-38ba-478b-b07c-a6aa36421caf","year":2023},{"cited_arxiv_id":"2107.03374","doi":"","is_internal_anchor":true,"ref_index":7,"title":"Evaluating Large Language Models Trained on Code","work_id":"042493e9-b26f-4b4e-bbde-382072ca9b08","year":2021},{"cited_arxiv_id":"2401.06066","doi":"10.48550/arxiv.2401.06066","is_internal_anchor":true,"ref_index":11,"title":"DeepSeekMoE: Towards Ultimate Expert Specialization in Mixture-of-Experts Language Models","work_id":"a9888d6d-bf47-4324-9834-7cc12ac3a78c","year":2024},{"cited_arxiv_id":"","doi":"","is_internal_anchor":false,"ref_index":12,"title":"T. Dao. Flash A ttention-2: Faster attention with better parallelism and work partitioning, 2023","work_id":"571162b3-cda9-4125-b3d2-5904014985c1","year":2023}],"snapshot_sha256":"7439d9d44296192850ae12f7699688a3c6cf2264fe0baa11a449a5910375d04c"},"source":{"id":"2405.04434","kind":"arxiv","version":5},"verdict":{"created_at":"2026-05-11T05:30:50.710829Z","id":"6bb7c623-af08-4124-8aec-ab0989e4dd26","model_set":{"reader":"grok-4.3"},"one_line_summary":"DeepSeek-V2 delivers top-tier open-source LLM performance using only 21B active parameters by compressing the KV cache 93.3% and cutting training costs 42.5% via MLA and DeepSeekMoE.","pipeline_version":"pith-pipeline@v0.9.0","pith_extraction_headline":"DeepSeek-V2 shows a Mixture-of-Experts model with 236 billion total parameters but only 21 billion activated per token can match top open-source language models while lowering training costs and inference demands.","strongest_claim":"even with only 21B activated parameters, DeepSeek-V2 and its chat versions still achieve top-tier performance among open-source models.","weakest_assumption":"That the reported performance and efficiency gains were measured against fair, standardized baselines without post-hoc data selection or undisclosed implementation advantages."}},"verdict_id":"6bb7c623-af08-4124-8aec-ab0989e4dd26"}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:08486a4f15e9e7c9f5e7d6e929fa0a4b6736dffbf9d7de542f78232f789457ae","target":"record","created_at":"2026-07-05T08:34:18Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"aa1f21fe687cbf7b5fce7a5b9ce802a45332d6122c30914c866eaf54ca837e83","cross_cats_sorted":["cs.AI"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-05-07T15:56:43Z","title_canon_sha256":"a4fbad70e2dec8122e1eb20bbd56a6f5df3fa6e0323eb665c0516c199914368f"},"schema_version":"1.0","source":{"id":"2405.04434","kind":"arxiv","version":5}},"canonical_sha256":"068ace48eab586856d8c2e36d7a3ad0c5ec333f6210d2271dd582973259b8c11","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"068ace48eab586856d8c2e36d7a3ad0c5ec333f6210d2271dd582973259b8c11","first_computed_at":"2026-07-05T08:34:18.327108Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T08:34:18.327108Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"tkjxqq3atgxP8LKobD54mZNVX2roNwVZTYH3cb8WSrWbKRZVc8CDlYeqgSdue5wWRfGJ6VBh+vaBveOW34pSCw==","signature_status":"signed_v1","signed_at":"2026-07-05T08:34:18.327687Z","signed_message":"canonical_sha256_bytes"},"source_id":"2405.04434","source_kind":"arxiv","source_version":5}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:08486a4f15e9e7c9f5e7d6e929fa0a4b6736dffbf9d7de542f78232f789457ae","sha256:de1e1b2cb874a02f5107bca90d03a446ceb72405be7ef49959a7702f0a272fbc"],"state_sha256":"eb65a2df10084ce42f65d7d765281a0141e8054aa252fdd0313ab9642fbdece6"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"vhJg3/yHCMeIl8XCyHnlgxpxI2/PO36H1hOpCH+y02MV5jlDE5SXDTqSRSpr0I+CfI902FIwalCiKu74R0MACA==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-12T16:45:04.416816Z","bundle_sha256":"730f5449847195d2e9c058c18704b352470a17e11de59084c3989ceda3edd79b"}}