{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2025:M57F6ZQWAJKCZSE3QJGDAA7KKR","short_pith_number":"pith:M57F6ZQW","canonical_record":{"source":{"id":"2506.13585","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2025-06-16T15:08:02Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"4df25fb124b3bfe8ba8091b0a12678bec2d38e8439cc546a1c573f36b7f0ec6f","abstract_canon_sha256":"199a92c30626619fdc852d97956e3eef1f555f139261c3c9209adcd022cc10b3"},"schema_version":"1.0"},"canonical_sha256":"677e5f661602542cc89b824c3003ea5443c7fc953ccaeaca4a3ff2f57bc965b3","source":{"kind":"arxiv","id":"2506.13585","version":1},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2506.13585","created_at":"2026-07-05T11:22:22Z"},{"alias_kind":"arxiv_version","alias_value":"2506.13585v1","created_at":"2026-07-05T11:22:22Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.13585","created_at":"2026-07-05T11:22:22Z"},{"alias_kind":"pith_short_12","alias_value":"M57F6ZQWAJKC","created_at":"2026-07-05T11:22:22Z"},{"alias_kind":"pith_short_16","alias_value":"M57F6ZQWAJKCZSE3","created_at":"2026-07-05T11:22:22Z"},{"alias_kind":"pith_short_8","alias_value":"M57F6ZQW","created_at":"2026-07-05T11:22:22Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2025:M57F6ZQWAJKCZSE3QJGDAA7KKR","target":"record","payload":{"canonical_record":{"source":{"id":"2506.13585","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2025-06-16T15:08:02Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"4df25fb124b3bfe8ba8091b0a12678bec2d38e8439cc546a1c573f36b7f0ec6f","abstract_canon_sha256":"199a92c30626619fdc852d97956e3eef1f555f139261c3c9209adcd022cc10b3"},"schema_version":"1.0"},"canonical_sha256":"677e5f661602542cc89b824c3003ea5443c7fc953ccaeaca4a3ff2f57bc965b3","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:22:22.837156Z","signature_b64":"xwmu4+S32u/X73y0mCIfL2IGbwGmL9ejXrGqi64YaeRZ+zIAMz+OwHDHDsB0naKPMltVIlUgC4ByXGT+j0MqBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"677e5f661602542cc89b824c3003ea5443c7fc953ccaeaca4a3ff2f57bc965b3","last_reissued_at":"2026-07-05T11:22:22.836714Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:22:22.836714Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2506.13585","source_version":1,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T11:22:22Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"AWbD5WoYS7Ce6iis4W8EhG3KYtrcIoKDYpaGggtY7mfVXweRiFGJrwa+qopg3LJx8x+woE4Cm+5xdRkB6hehDg==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-04T02:45:26.216290Z"},"content_sha256":"114ee3a16bbb9a363229bc24d0fe9383c34d0aaeea99b54f40d6a798dd6fca03","schema_version":"1.0","event_id":"sha256:114ee3a16bbb9a363229bc24d0fe9383c34d0aaeea99b54f40d6a798dd6fca03"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2025:M57F6ZQWAJKCZSE3QJGDAA7KKR","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"MiniMax-M1: Scaling Test-Time Compute Efficiently with Lightning Attention","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"MiniMax-M1 combines hybrid attention with a new RL algorithm to scale test-time compute efficiently in a 456 billion parameter model.","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Aonian Li, Bangwei Gong, Binyang Jiang, Bo Fei, Boji Shan, Bo Yang, Changqing Yu, Chao Wang, Chengjun Xiao, Chengyu Du, Cheng Zhu, Chi Zhang, Chunhao Zhang, Chunhui Du, Chu Qiao, Congchao Guo, Da Chen, Deming Ding, Dianjun Sun, Dong Li, Enwei Jiao, Haichao Zhu, Haigang Zhou, Haimo Zhang, Han Ding, Haohai Sun, Haoyu Feng, Huaiguang Cai, Jian Sun, Jiaqi Zhuang, Jiaren Cai, Jiayuan Song, Jingyang Li, Jinhao Tian, Jinli Liu, Jin Zhu, Junhao Xu, Junjie Yan, Junteng Liu, Junxian He, Kaiyi Feng, Kecheng Xiao, Ke Yang, Le Han, Leyang Wang, Lianfei Yu, Liheng Feng, Linge Du, Lingyu Yang, Lin Li, Lin Zheng, Lunbin Zeng, Minghui Yu, Mingliang Tao, Mingyuan Chi, MiniMax: Aili Chen, Mozhi Zhang, Mujie Lin, Nan Hu, Nongyu Di, Pengfei Li, Peng Gao, Pengyu Zhao, Qibing Ren, Qidi Xu, Qile Li, Qin Wang, Rong Tian, Ruitao Leng, Shaoxiang Chen, Shaoyu Chen, Shengmin Shi, Shitong Weng, Shuchang Guan, Shuqi Yu, Sichen Li, Songquan Zhu, Tengfei Li, Tianchi Cai, Tianrun Liang, Weiyu Cheng, Weize Kong, Wenkai Li, Xiancai Chen, Xiangjun Song, Xiaobo Li, Xiaodong Han, Xiao Luo, Xiao Su, Xinzhu Hou, Xuan Lu, Xun Zou, Xuyang Shen, Yan Gong, Yang Wang, Yan Ma, Yiqi Shi, Yiran Zhong, Yonghong Duan, Yongxiang Fu, Yongyi Hu, Yuanxiang Fan, Yufeng Yang, Yu Gao, Yuhao Li, Yulin Hu, Yunan Huang, Yunji Li, Yunzhi Xu, Yuxin Mao, Yuxuan Shi, Yuze Wenren, Zehan Li, Zelin Li, Zhanxu Tian, Zhengmao Zhu, Zhenhua Fan, Zhenzhen Wu, Zhichao Xu, Zhihang Yu, Zhiheng Lyu, Zhuo Jiang, Zibo Gao, Zijian Song, Zijia Wu, Zijun Sun","submitted_at":"2025-06-16T15:08:02Z","abstract_excerpt":"We introduce MiniMax-M1, the world's first open-weight, large-scale hybrid-attention reasoning model. MiniMax-M1 is powered by a hybrid Mixture-of-Experts (MoE) architecture combined with a lightning attention mechanism. The model is developed based on our previous MiniMax-Text-01 model, which contains a total of 456 billion parameters with 45.9 billion parameters activated per token. The M1 model natively supports a context length of 1 million tokens, 8x the context size of DeepSeek R1. Furthermore, the lightning attention mechanism in MiniMax-M1 enables efficient scaling of test-time compute"},"claims":{"count":4,"items":[{"kind":"strongest_claim","text":"MiniMax-M1 is the world's first open-weight, large-scale hybrid-attention reasoning model... Experiments on standard benchmarks show that our models are comparable or superior to strong open-weight models such as the original DeepSeek-R1 and Qwen3-235B, with particular strengths in complex software engineering, tool utilization, and long-context tasks.","source":"verdict.strongest_claim","status":"machine_extracted","claim_id":"C1","attestation":"unclaimed"},{"kind":"weakest_assumption","text":"The assumption that the lightning attention mechanism enables efficient scaling of test-time compute and that the CISPO algorithm outperforms other RL variants without introducing biases or performance losses.","source":"verdict.weakest_assumption","status":"machine_extracted","claim_id":"C2","attestation":"unclaimed"},{"kind":"one_line_summary","text":"MiniMax-M1 is a 456B parameter hybrid-attention MoE model trained with CISPO RL that achieves performance comparable or superior to DeepSeek-R1 and Qwen3-235B on reasoning and software engineering tasks while training in three weeks on 512 GPUs.","source":"verdict.one_line_summary","status":"machine_extracted","claim_id":"C3","attestation":"unclaimed"},{"kind":"headline","text":"MiniMax-M1 combines hybrid attention with a new RL algorithm to scale test-time compute efficiently in a 456 billion parameter model.","source":"verdict.pith_extraction.headline","status":"machine_extracted","claim_id":"C4","attestation":"unclaimed"}],"snapshot_sha256":"406207101785708b766413969083e51f412c92239f2d69b3f708ed7feb2304d3"},"source":{"id":"2506.13585","kind":"arxiv","version":1},"verdict":{"id":"ba63c258-ff57-428d-a9d7-43163e3f0371","model_set":{"reader":"grok-4.3"},"created_at":"2026-05-12T09:22:33.467728Z","strongest_claim":"MiniMax-M1 is the world's first open-weight, large-scale hybrid-attention reasoning model... Experiments on standard benchmarks show that our models are comparable or superior to strong open-weight models such as the original DeepSeek-R1 and Qwen3-235B, with particular strengths in complex software engineering, tool utilization, and long-context tasks.","one_line_summary":"MiniMax-M1 is a 456B parameter hybrid-attention MoE model trained with CISPO RL that achieves performance comparable or superior to DeepSeek-R1 and Qwen3-235B on reasoning and software engineering tasks while training in three weeks on 512 GPUs.","pipeline_version":"pith-pipeline@v0.9.0","weakest_assumption":"The assumption that the lightning attention mechanism enables efficient scaling of test-time compute and that the CISPO algorithm outperforms other RL variants without introducing biases or performance losses.","pith_extraction_headline":"MiniMax-M1 combines hybrid attention with a new RL algorithm to scale test-time compute efficiently in a 456 billion parameter model."},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2506.13585/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":50,"sample":[{"doi":"","year":null,"title":"Simple linear attention language models balance the recall-throughput tradeoff","work_id":"c542bc4f-de21-42d2-812b-b46f5fa9b434","ref_index":1,"cited_arxiv_id":"","is_internal_anchor":false},{"doi":"","year":null,"title":"LongBench v2: Towards Deeper Understanding and Reasoning on Realistic Long-context Multitasks","work_id":"9fac250b-241e-41ce-9177-469deaf03040","ref_index":2,"cited_arxiv_id":"2412.15204","is_internal_anchor":true},{"doi":"","year":null,"title":"Titans: Learning to Memorize at Test Time","work_id":"fb2b7625-b733-43cb-af52-00b0a31a8d7f","ref_index":3,"cited_arxiv_id":"2501.00663","is_internal_anchor":true},{"doi":"","year":2004,"title":"Longformer: The Long-Document Transformer","work_id":"abea7a44-6668-4de7-aab6-f53a6e5aa088","ref_index":4,"cited_arxiv_id":"2004.05150","is_internal_anchor":true},{"doi":"","year":null,"title":"Empirical Evaluation of Gated Recurrent Neural Networks on Sequence Modeling","work_id":"c7f2f5a9-ae4b-48db-aff0-24b9d0528995","ref_index":5,"cited_arxiv_id":"1412.3555","is_internal_anchor":true}],"resolved_work":50,"snapshot_sha256":"b74f71c1115602c15bbb1702af04ad6f0c53eb989b08f0c25253633d9b33933a","internal_anchors":29},"formal_canon":{"evidence_count":2,"snapshot_sha256":"9c2ff16a9c216f0f442d9c6b4f5e560a6e019edb2591f1540135c6e9b0a0a01a"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":"ba63c258-ff57-428d-a9d7-43163e3f0371"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T11:22:22Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"xz3rGqr4eeyzgc1cK5p4B46hiTBv1GcUIZcI/VX+3ejGuhCDL0xHjwQGYRJbnXKMiWLBfB/3T22AuM0jqqG7CA==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-04T02:45:26.217468Z"},"content_sha256":"2df350d5bf5dce0515d3b26239d395b71f9034d04aa1d76eacca015f956214a8","schema_version":"1.0","event_id":"sha256:2df350d5bf5dce0515d3b26239d395b71f9034d04aa1d76eacca015f956214a8"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/M57F6ZQWAJKCZSE3QJGDAA7KKR/bundle.json","state_url":"https://pith.science/pith/M57F6ZQWAJKCZSE3QJGDAA7KKR/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/M57F6ZQWAJKCZSE3QJGDAA7KKR/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-04T02:45:26Z","links":{"resolver":"https://pith.science/pith/M57F6ZQWAJKCZSE3QJGDAA7KKR","bundle":"https://pith.science/pith/M57F6ZQWAJKCZSE3QJGDAA7KKR/bundle.json","state":"https://pith.science/pith/M57F6ZQWAJKCZSE3QJGDAA7KKR/state.json","well_known_bundle":"https://pith.science/.well-known/pith/M57F6ZQWAJKCZSE3QJGDAA7KKR/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2025:M57F6ZQWAJKCZSE3QJGDAA7KKR","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"199a92c30626619fdc852d97956e3eef1f555f139261c3c9209adcd022cc10b3","cross_cats_sorted":["cs.LG"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2025-06-16T15:08:02Z","title_canon_sha256":"4df25fb124b3bfe8ba8091b0a12678bec2d38e8439cc546a1c573f36b7f0ec6f"},"schema_version":"1.0","source":{"id":"2506.13585","kind":"arxiv","version":1}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2506.13585","created_at":"2026-07-05T11:22:22Z"},{"alias_kind":"arxiv_version","alias_value":"2506.13585v1","created_at":"2026-07-05T11:22:22Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.13585","created_at":"2026-07-05T11:22:22Z"},{"alias_kind":"pith_short_12","alias_value":"M57F6ZQWAJKC","created_at":"2026-07-05T11:22:22Z"},{"alias_kind":"pith_short_16","alias_value":"M57F6ZQWAJKCZSE3","created_at":"2026-07-05T11:22:22Z"},{"alias_kind":"pith_short_8","alias_value":"M57F6ZQW","created_at":"2026-07-05T11:22:22Z"}],"graph_snapshots":[{"event_id":"sha256:2df350d5bf5dce0515d3b26239d395b71f9034d04aa1d76eacca015f956214a8","target":"graph","created_at":"2026-07-05T11:22:22Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":4,"items":[{"attestation":"unclaimed","claim_id":"C1","kind":"strongest_claim","source":"verdict.strongest_claim","status":"machine_extracted","text":"MiniMax-M1 is the world's first open-weight, large-scale hybrid-attention reasoning model... Experiments on standard benchmarks show that our models are comparable or superior to strong open-weight models such as the original DeepSeek-R1 and Qwen3-235B, with particular strengths in complex software engineering, tool utilization, and long-context tasks."},{"attestation":"unclaimed","claim_id":"C2","kind":"weakest_assumption","source":"verdict.weakest_assumption","status":"machine_extracted","text":"The assumption that the lightning attention mechanism enables efficient scaling of test-time compute and that the CISPO algorithm outperforms other RL variants without introducing biases or performance losses."},{"attestation":"unclaimed","claim_id":"C3","kind":"one_line_summary","source":"verdict.one_line_summary","status":"machine_extracted","text":"MiniMax-M1 is a 456B parameter hybrid-attention MoE model trained with CISPO RL that achieves performance comparable or superior to DeepSeek-R1 and Qwen3-235B on reasoning and software engineering tasks while training in three weeks on 512 GPUs."},{"attestation":"unclaimed","claim_id":"C4","kind":"headline","source":"verdict.pith_extraction.headline","status":"machine_extracted","text":"MiniMax-M1 combines hybrid attention with a new RL algorithm to scale test-time compute efficiently in a 456 billion parameter model."}],"snapshot_sha256":"406207101785708b766413969083e51f412c92239f2d69b3f708ed7feb2304d3"},"formal_canon":{"evidence_count":2,"snapshot_sha256":"9c2ff16a9c216f0f442d9c6b4f5e560a6e019edb2591f1540135c6e9b0a0a01a"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2506.13585/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"We introduce MiniMax-M1, the world's first open-weight, large-scale hybrid-attention reasoning model. MiniMax-M1 is powered by a hybrid Mixture-of-Experts (MoE) architecture combined with a lightning attention mechanism. The model is developed based on our previous MiniMax-Text-01 model, which contains a total of 456 billion parameters with 45.9 billion parameters activated per token. The M1 model natively supports a context length of 1 million tokens, 8x the context size of DeepSeek R1. Furthermore, the lightning attention mechanism in MiniMax-M1 enables efficient scaling of test-time compute","authors_text":"Aonian Li, Bangwei Gong, Binyang Jiang, Bo Fei, Boji Shan, Bo Yang, Changqing Yu, Chao Wang, Chengjun Xiao, Chengyu Du, Cheng Zhu, Chi Zhang, Chunhao Zhang, Chunhui Du, Chu Qiao, Congchao Guo, Da Chen, Deming Ding, Dianjun Sun, Dong Li, Enwei Jiao, Haichao Zhu, Haigang Zhou, Haimo Zhang, Han Ding, Haohai Sun, Haoyu Feng, Huaiguang Cai, Jian Sun, Jiaqi Zhuang, Jiaren Cai, Jiayuan Song, Jingyang Li, Jinhao Tian, Jinli Liu, Jin Zhu, Junhao Xu, Junjie Yan, Junteng Liu, Junxian He, Kaiyi Feng, Kecheng Xiao, Ke Yang, Le Han, Leyang Wang, Lianfei Yu, Liheng Feng, Linge Du, Lingyu Yang, Lin Li, Lin Zheng, Lunbin Zeng, Minghui Yu, Mingliang Tao, Mingyuan Chi, MiniMax: Aili Chen, Mozhi Zhang, Mujie Lin, Nan Hu, Nongyu Di, Pengfei Li, Peng Gao, Pengyu Zhao, Qibing Ren, Qidi Xu, Qile Li, Qin Wang, Rong Tian, Ruitao Leng, Shaoxiang Chen, Shaoyu Chen, Shengmin Shi, Shitong Weng, Shuchang Guan, Shuqi Yu, Sichen Li, Songquan Zhu, Tengfei Li, Tianchi Cai, Tianrun Liang, Weiyu Cheng, Weize Kong, Wenkai Li, Xiancai Chen, Xiangjun Song, Xiaobo Li, Xiaodong Han, Xiao Luo, Xiao Su, Xinzhu Hou, Xuan Lu, Xun Zou, Xuyang Shen, Yan Gong, Yang Wang, Yan Ma, Yiqi Shi, Yiran Zhong, Yonghong Duan, Yongxiang Fu, Yongyi Hu, Yuanxiang Fan, Yufeng Yang, Yu Gao, Yuhao Li, Yulin Hu, Yunan Huang, Yunji Li, Yunzhi Xu, Yuxin Mao, Yuxuan Shi, Yuze Wenren, Zehan Li, Zelin Li, Zhanxu Tian, Zhengmao Zhu, Zhenhua Fan, Zhenzhen Wu, Zhichao Xu, Zhihang Yu, Zhiheng Lyu, Zhuo Jiang, Zibo Gao, Zijian Song, Zijia Wu, Zijun Sun","cross_cats":["cs.LG"],"headline":"MiniMax-M1 combines hybrid attention with a new RL algorithm to scale test-time compute efficiently in a 456 billion parameter model.","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2025-06-16T15:08:02Z","title":"MiniMax-M1: Scaling Test-Time Compute Efficiently with Lightning Attention"},"references":{"count":50,"internal_anchors":29,"resolved_work":50,"sample":[{"cited_arxiv_id":"","doi":"","is_internal_anchor":false,"ref_index":1,"title":"Simple linear attention language models balance the recall-throughput tradeoff","work_id":"c542bc4f-de21-42d2-812b-b46f5fa9b434","year":null},{"cited_arxiv_id":"2412.15204","doi":"","is_internal_anchor":true,"ref_index":2,"title":"LongBench v2: Towards Deeper Understanding and Reasoning on Realistic Long-context Multitasks","work_id":"9fac250b-241e-41ce-9177-469deaf03040","year":null},{"cited_arxiv_id":"2501.00663","doi":"","is_internal_anchor":true,"ref_index":3,"title":"Titans: Learning to Memorize at Test Time","work_id":"fb2b7625-b733-43cb-af52-00b0a31a8d7f","year":null},{"cited_arxiv_id":"2004.05150","doi":"","is_internal_anchor":true,"ref_index":4,"title":"Longformer: The Long-Document Transformer","work_id":"abea7a44-6668-4de7-aab6-f53a6e5aa088","year":2004},{"cited_arxiv_id":"1412.3555","doi":"","is_internal_anchor":true,"ref_index":5,"title":"Empirical Evaluation of Gated Recurrent Neural Networks on Sequence Modeling","work_id":"c7f2f5a9-ae4b-48db-aff0-24b9d0528995","year":null}],"snapshot_sha256":"b74f71c1115602c15bbb1702af04ad6f0c53eb989b08f0c25253633d9b33933a"},"source":{"id":"2506.13585","kind":"arxiv","version":1},"verdict":{"created_at":"2026-05-12T09:22:33.467728Z","id":"ba63c258-ff57-428d-a9d7-43163e3f0371","model_set":{"reader":"grok-4.3"},"one_line_summary":"MiniMax-M1 is a 456B parameter hybrid-attention MoE model trained with CISPO RL that achieves performance comparable or superior to DeepSeek-R1 and Qwen3-235B on reasoning and software engineering tasks while training in three weeks on 512 GPUs.","pipeline_version":"pith-pipeline@v0.9.0","pith_extraction_headline":"MiniMax-M1 combines hybrid attention with a new RL algorithm to scale test-time compute efficiently in a 456 billion parameter model.","strongest_claim":"MiniMax-M1 is the world's first open-weight, large-scale hybrid-attention reasoning model... Experiments on standard benchmarks show that our models are comparable or superior to strong open-weight models such as the original DeepSeek-R1 and Qwen3-235B, with particular strengths in complex software engineering, tool utilization, and long-context tasks.","weakest_assumption":"The assumption that the lightning attention mechanism enables efficient scaling of test-time compute and that the CISPO algorithm outperforms other RL variants without introducing biases or performance losses."}},"verdict_id":"ba63c258-ff57-428d-a9d7-43163e3f0371"}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:114ee3a16bbb9a363229bc24d0fe9383c34d0aaeea99b54f40d6a798dd6fca03","target":"record","created_at":"2026-07-05T11:22:22Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"199a92c30626619fdc852d97956e3eef1f555f139261c3c9209adcd022cc10b3","cross_cats_sorted":["cs.LG"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2025-06-16T15:08:02Z","title_canon_sha256":"4df25fb124b3bfe8ba8091b0a12678bec2d38e8439cc546a1c573f36b7f0ec6f"},"schema_version":"1.0","source":{"id":"2506.13585","kind":"arxiv","version":1}},"canonical_sha256":"677e5f661602542cc89b824c3003ea5443c7fc953ccaeaca4a3ff2f57bc965b3","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"677e5f661602542cc89b824c3003ea5443c7fc953ccaeaca4a3ff2f57bc965b3","first_computed_at":"2026-07-05T11:22:22.836714Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T11:22:22.836714Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"xwmu4+S32u/X73y0mCIfL2IGbwGmL9ejXrGqi64YaeRZ+zIAMz+OwHDHDsB0naKPMltVIlUgC4ByXGT+j0MqBw==","signature_status":"signed_v1","signed_at":"2026-07-05T11:22:22.837156Z","signed_message":"canonical_sha256_bytes"},"source_id":"2506.13585","source_kind":"arxiv","source_version":1}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:114ee3a16bbb9a363229bc24d0fe9383c34d0aaeea99b54f40d6a798dd6fca03","sha256:2df350d5bf5dce0515d3b26239d395b71f9034d04aa1d76eacca015f956214a8"],"state_sha256":"2dc8bb1dae17ac348133102569a80be6a10d33e07027cdf2ca8e6979b673b6a8"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"uSLSpP1h0Xfu3Anc2H8Ng1CUbXyunYKyCzvo4vR8xOqXlQbT/1HshfUaMhTCiIAjx5B0e8Pr+c4KzrVpEsVDCQ==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-04T02:45:26.222100Z","bundle_sha256":"ec12d7c791b973637a505fe8e570b352c2dc91dd6aa9fcd631599d0415d88d43"}}