{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2024:FQZSKWREF54DN23K6NGDRENPHG","short_pith_number":"pith:FQZSKWRE","canonical_record":{"source":{"id":"2406.12793","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-06-18T16:58:21Z","cross_cats_sorted":[],"title_canon_sha256":"7e5cbb2258df8a8cea4f770a069588bd123929839363f246b2f828e70f7aed5f","abstract_canon_sha256":"e1d79a50f6830c2440d1a199ae84f45637fb340943fa5dd9637533d3a00c2319"},"schema_version":"1.0"},"canonical_sha256":"2c33255a242f7836eb6af34c3891af39b025fa421221e48c93923671fa6c20f9","source":{"kind":"arxiv","id":"2406.12793","version":2},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2406.12793","created_at":"2026-07-05T08:49:53Z"},{"alias_kind":"arxiv_version","alias_value":"2406.12793v2","created_at":"2026-07-05T08:49:53Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.12793","created_at":"2026-07-05T08:49:53Z"},{"alias_kind":"pith_short_12","alias_value":"FQZSKWREF54D","created_at":"2026-07-05T08:49:53Z"},{"alias_kind":"pith_short_16","alias_value":"FQZSKWREF54DN23K","created_at":"2026-07-05T08:49:53Z"},{"alias_kind":"pith_short_8","alias_value":"FQZSKWRE","created_at":"2026-07-05T08:49:53Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2024:FQZSKWREF54DN23K6NGDRENPHG","target":"record","payload":{"canonical_record":{"source":{"id":"2406.12793","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-06-18T16:58:21Z","cross_cats_sorted":[],"title_canon_sha256":"7e5cbb2258df8a8cea4f770a069588bd123929839363f246b2f828e70f7aed5f","abstract_canon_sha256":"e1d79a50f6830c2440d1a199ae84f45637fb340943fa5dd9637533d3a00c2319"},"schema_version":"1.0"},"canonical_sha256":"2c33255a242f7836eb6af34c3891af39b025fa421221e48c93923671fa6c20f9","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:49:53.727038Z","signature_b64":"8zP5cEdGXj62SFU8q4xtT58yrtPWweIzZ3B3NmrcONzP0f2ryCpkUDLmnbdDs61l36MH2ijtdarQxnt4THt4CQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"2c33255a242f7836eb6af34c3891af39b025fa421221e48c93923671fa6c20f9","last_reissued_at":"2026-07-05T08:49:53.726540Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:49:53.726540Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"2406.12793","source_version":2,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T08:49:53Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"yCn8xQe22ROP+VGyiv8G440dwORA5tP6DCoQUsD7DuosQ0cHB6xWhM+ourwww1Cp1vXTkfUb1v+xnWsAL59JDw==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-12T16:42:03.331611Z"},"content_sha256":"db277c3f9737c5034eebde11569e670040bdc2e8f9985edb04bffcf666dac360","schema_version":"1.0","event_id":"sha256:db277c3f9737c5034eebde11569e670040bdc2e8f9985edb04bffcf666dac360"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2024:FQZSKWREF54DN23K6NGDRENPHG","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"ChatGLM: A Family of Large Language Models from GLM-130B to GLM-4 All Tools","license":"http://creativecommons.org/licenses/by/4.0/","headline":"GLM-4 language models rival or surpass GPT-4 on benchmarks for general ability, reasoning, coding, and Chinese alignment.","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Bin Xu, Bowen Wang, Chenhui Zhang, Dan Zhang, Da Yin, Diego Rojas, Guanyu Feng, Hanlin Zhao, Hanyu Lai, Hao Yu, Hongning Wang, Jiadai Sun, Jiajie Zhang, Jiale Cheng, Jiayi Gui, Jie Tang, Jingyu Sun, Jing Zhang, Juanzi Li, Lei Zhao, Lindong Wu, Lucen Zhong, Mingdao Liu, Minlie Huang, Peng Zhang, Qinkai Zheng, Rui Lu, Shuaiqi Duan, Shudan Zhang, Shulin Cao, Shuxun Yang, Team GLM: Aohan Zeng, Weng Lam Tam, Wenyi Zhao, Xiaohan Zhang, Xiao Liu, Xiaotao Gu, Xiao Xia, Xinghan Liu, Xin Lv, Xinyi Liu, Xinyue Yang, Xixuan Song, Xunkai Zhang, Yifan An, Yifan Xu, Yilin Niu, Yuantao Yang, Yueyan Li, Yushi Bai, Yuxiao Dong, Zehan Qi, Zhaoyu Wang, Zhengxiao Du, Zhen Yang, Zhenyu Hou, Zihan Wang","submitted_at":"2024-06-18T16:58:21Z","abstract_excerpt":"We introduce ChatGLM, an evolving family of large language models that we have been developing over time. This report primarily focuses on the GLM-4 language series, which includes GLM-4, GLM-4-Air, and GLM-4-9B. They represent our most capable models that are trained with all the insights and lessons gained from the preceding three generations of ChatGLM. To date, the GLM-4 models are pre-trained on ten trillions of tokens mostly in Chinese and English, along with a small set of corpus from 24 languages, and aligned primarily for Chinese and English usage. The high-quality alignment is achiev"},"claims":{"count":4,"items":[{"kind":"strongest_claim","text":"Evaluations show that GLM-4 1) closely rivals or outperforms GPT-4 in terms of general metrics such as MMLU, GSM8K, MATH, BBH, GPQA, and HumanEval, 2) gets close to GPT-4-Turbo in instruction following as measured by IFEval, 3) matches GPT-4 Turbo (128K) and Claude 3 for long context tasks, and 4) outperforms GPT-4 in Chinese alignments as measured by AlignBench.","source":"verdict.strongest_claim","status":"machine_extracted","claim_id":"C1","attestation":"unclaimed"},{"kind":"weakest_assumption","text":"That the reported benchmark scores reflect genuine capability rather than test-set contamination, prompt engineering, or selective reporting, given that full training data, exact evaluation protocols, and GLM-4 model weights are not released in the report.","source":"verdict.weakest_assumption","status":"machine_extracted","claim_id":"C2","attestation":"unclaimed"},{"kind":"one_line_summary","text":"GLM-4 models rival or exceed GPT-4 on MMLU, GSM8K, MATH, BBH, GPQA, HumanEval, IFEval, long-context tasks, and Chinese alignment while adding autonomous tool use for web, code, and image generation.","source":"verdict.one_line_summary","status":"machine_extracted","claim_id":"C3","attestation":"unclaimed"},{"kind":"headline","text":"GLM-4 language models rival or surpass GPT-4 on benchmarks for general ability, reasoning, coding, and Chinese alignment.","source":"verdict.pith_extraction.headline","status":"machine_extracted","claim_id":"C4","attestation":"unclaimed"}],"snapshot_sha256":"ac30b0efda64e8b2bacf97a6812b46ab2240789750e552533f291b3e20d640e8"},"source":{"id":"2406.12793","kind":"arxiv","version":2},"verdict":{"id":"b5a7128a-a712-474f-ab7d-f6dfff956612","model_set":{"reader":"grok-4.3"},"created_at":"2026-05-11T08:01:06.427306Z","strongest_claim":"Evaluations show that GLM-4 1) closely rivals or outperforms GPT-4 in terms of general metrics such as MMLU, GSM8K, MATH, BBH, GPQA, and HumanEval, 2) gets close to GPT-4-Turbo in instruction following as measured by IFEval, 3) matches GPT-4 Turbo (128K) and Claude 3 for long context tasks, and 4) outperforms GPT-4 in Chinese alignments as measured by AlignBench.","one_line_summary":"GLM-4 models rival or exceed GPT-4 on MMLU, GSM8K, MATH, BBH, GPQA, HumanEval, IFEval, long-context tasks, and Chinese alignment while adding autonomous tool use for web, code, and image generation.","pipeline_version":"pith-pipeline@v0.9.0","weakest_assumption":"That the reported benchmark scores reflect genuine capability rather than test-set contamination, prompt engineering, or selective reporting, given that full training data, exact evaluation protocols, and GLM-4 model weights are not released in the report.","pith_extraction_headline":"GLM-4 language models rival or surpass GPT-4 on benchmarks for general ability, reasoning, coding, and Chinese alignment."},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.12793/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":62,"sample":[{"doi":"","year":2024,"title":"Y . Bai, X. Lv, J. Zhang, Y . He, J. Qi, L. Hou, J. Tang, Y . Dong, and J. Li. Longalign: A recipe for long context alignment of large language models, 2024","work_id":"934bd228-b897-46f9-a7e5-2ccd04ee2a7d","ref_index":1,"cited_arxiv_id":"","is_internal_anchor":false},{"doi":"","year":2023,"title":"Y . Bai, X. Lv, J. Zhang, H. Lyu, J. Tang, Z. Huang, Z. Du, X. Liu, A. Zeng, L. Hou, Y . Dong, J. Tang, and J. Li. Longbench: A bilingual, multitask benchmark for long context understanding, 2023","work_id":"48cbb890-8e16-4eb1-8f1a-de441b4acb71","ref_index":2,"cited_arxiv_id":"","is_internal_anchor":false},{"doi":"","year":2020,"title":"T. B. Brown, B. Mann, N. Ryder, M. Subbiah, J. Kaplan, P. Dhariwal, A. Neelakantan, P. Shyam, G. Sastry, A. Askell, S. Agarwal, A. Herbert-V oss, G. Krueger, T. Henighan, R. Child, A. Ramesh, D. M. Zi","work_id":"46235657-731f-4001-b42a-8f57527cb375","ref_index":3,"cited_arxiv_id":"","is_internal_anchor":false},{"doi":"","year":2021,"title":"Evaluating Large Language Models Trained on Code","work_id":"042493e9-b26f-4b4e-bbde-382072ca9b08","ref_index":4,"cited_arxiv_id":"2107.03374","is_internal_anchor":true},{"doi":"","year":2023,"title":"Extending Context Window of Large Language Models via Positional Interpolation","work_id":"c8b6df85-e7da-4bd8-90a4-d309cc2a0f60","ref_index":5,"cited_arxiv_id":"2306.15595","is_internal_anchor":true}],"resolved_work":62,"snapshot_sha256":"408a97c7b99f1b8bc4c9ea667a204c525ed44b53e7dff1faae66efb2e5a084ca","internal_anchors":15},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":"b5a7128a-a712-474f-ab7d-f6dfff956612"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-07-05T08:49:53Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"ym+tNNW2QSXx9nIQC5AXFZ1wSLIWmkN+mgvTsAtHVX2kb0oNdWMzOU180UcuzUwmjc7C5yPWkoM7DAYCJbUNDg==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-08-12T16:42:03.332819Z"},"content_sha256":"703efb3a46191414549d14f1d76776662271398e9324a616f18710c37bbd3086","schema_version":"1.0","event_id":"sha256:703efb3a46191414549d14f1d76776662271398e9324a616f18710c37bbd3086"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/FQZSKWREF54DN23K6NGDRENPHG/bundle.json","state_url":"https://pith.science/pith/FQZSKWREF54DN23K6NGDRENPHG/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/FQZSKWREF54DN23K6NGDRENPHG/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-08-12T16:42:03Z","links":{"resolver":"https://pith.science/pith/FQZSKWREF54DN23K6NGDRENPHG","bundle":"https://pith.science/pith/FQZSKWREF54DN23K6NGDRENPHG/bundle.json","state":"https://pith.science/pith/FQZSKWREF54DN23K6NGDRENPHG/state.json","well_known_bundle":"https://pith.science/.well-known/pith/FQZSKWREF54DN23K6NGDRENPHG/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2024:FQZSKWREF54DN23K6NGDRENPHG","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"e1d79a50f6830c2440d1a199ae84f45637fb340943fa5dd9637533d3a00c2319","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-06-18T16:58:21Z","title_canon_sha256":"7e5cbb2258df8a8cea4f770a069588bd123929839363f246b2f828e70f7aed5f"},"schema_version":"1.0","source":{"id":"2406.12793","kind":"arxiv","version":2}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"2406.12793","created_at":"2026-07-05T08:49:53Z"},{"alias_kind":"arxiv_version","alias_value":"2406.12793v2","created_at":"2026-07-05T08:49:53Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.12793","created_at":"2026-07-05T08:49:53Z"},{"alias_kind":"pith_short_12","alias_value":"FQZSKWREF54D","created_at":"2026-07-05T08:49:53Z"},{"alias_kind":"pith_short_16","alias_value":"FQZSKWREF54DN23K","created_at":"2026-07-05T08:49:53Z"},{"alias_kind":"pith_short_8","alias_value":"FQZSKWRE","created_at":"2026-07-05T08:49:53Z"}],"graph_snapshots":[{"event_id":"sha256:703efb3a46191414549d14f1d76776662271398e9324a616f18710c37bbd3086","target":"graph","created_at":"2026-07-05T08:49:53Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":4,"items":[{"attestation":"unclaimed","claim_id":"C1","kind":"strongest_claim","source":"verdict.strongest_claim","status":"machine_extracted","text":"Evaluations show that GLM-4 1) closely rivals or outperforms GPT-4 in terms of general metrics such as MMLU, GSM8K, MATH, BBH, GPQA, and HumanEval, 2) gets close to GPT-4-Turbo in instruction following as measured by IFEval, 3) matches GPT-4 Turbo (128K) and Claude 3 for long context tasks, and 4) outperforms GPT-4 in Chinese alignments as measured by AlignBench."},{"attestation":"unclaimed","claim_id":"C2","kind":"weakest_assumption","source":"verdict.weakest_assumption","status":"machine_extracted","text":"That the reported benchmark scores reflect genuine capability rather than test-set contamination, prompt engineering, or selective reporting, given that full training data, exact evaluation protocols, and GLM-4 model weights are not released in the report."},{"attestation":"unclaimed","claim_id":"C3","kind":"one_line_summary","source":"verdict.one_line_summary","status":"machine_extracted","text":"GLM-4 models rival or exceed GPT-4 on MMLU, GSM8K, MATH, BBH, GPQA, HumanEval, IFEval, long-context tasks, and Chinese alignment while adding autonomous tool use for web, code, and image generation."},{"attestation":"unclaimed","claim_id":"C4","kind":"headline","source":"verdict.pith_extraction.headline","status":"machine_extracted","text":"GLM-4 language models rival or surpass GPT-4 on benchmarks for general ability, reasoning, coding, and Chinese alignment."}],"snapshot_sha256":"ac30b0efda64e8b2bacf97a6812b46ab2240789750e552533f291b3e20d640e8"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"integrity":{"available":true,"clean":true,"detectors_run":[],"endpoint":"/pith/2406.12793/integrity.json","findings":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938","summary":{"advisory":0,"by_detector":{},"critical":0,"informational":0}},"paper":{"abstract_excerpt":"We introduce ChatGLM, an evolving family of large language models that we have been developing over time. This report primarily focuses on the GLM-4 language series, which includes GLM-4, GLM-4-Air, and GLM-4-9B. They represent our most capable models that are trained with all the insights and lessons gained from the preceding three generations of ChatGLM. To date, the GLM-4 models are pre-trained on ten trillions of tokens mostly in Chinese and English, along with a small set of corpus from 24 languages, and aligned primarily for Chinese and English usage. The high-quality alignment is achiev","authors_text":"Bin Xu, Bowen Wang, Chenhui Zhang, Dan Zhang, Da Yin, Diego Rojas, Guanyu Feng, Hanlin Zhao, Hanyu Lai, Hao Yu, Hongning Wang, Jiadai Sun, Jiajie Zhang, Jiale Cheng, Jiayi Gui, Jie Tang, Jingyu Sun, Jing Zhang, Juanzi Li, Lei Zhao, Lindong Wu, Lucen Zhong, Mingdao Liu, Minlie Huang, Peng Zhang, Qinkai Zheng, Rui Lu, Shuaiqi Duan, Shudan Zhang, Shulin Cao, Shuxun Yang, Team GLM: Aohan Zeng, Weng Lam Tam, Wenyi Zhao, Xiaohan Zhang, Xiao Liu, Xiaotao Gu, Xiao Xia, Xinghan Liu, Xin Lv, Xinyi Liu, Xinyue Yang, Xixuan Song, Xunkai Zhang, Yifan An, Yifan Xu, Yilin Niu, Yuantao Yang, Yueyan Li, Yushi Bai, Yuxiao Dong, Zehan Qi, Zhaoyu Wang, Zhengxiao Du, Zhen Yang, Zhenyu Hou, Zihan Wang","cross_cats":[],"headline":"GLM-4 language models rival or surpass GPT-4 on benchmarks for general ability, reasoning, coding, and Chinese alignment.","license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-06-18T16:58:21Z","title":"ChatGLM: A Family of Large Language Models from GLM-130B to GLM-4 All Tools"},"references":{"count":62,"internal_anchors":15,"resolved_work":62,"sample":[{"cited_arxiv_id":"","doi":"","is_internal_anchor":false,"ref_index":1,"title":"Y . Bai, X. Lv, J. Zhang, Y . He, J. Qi, L. Hou, J. Tang, Y . Dong, and J. Li. Longalign: A recipe for long context alignment of large language models, 2024","work_id":"934bd228-b897-46f9-a7e5-2ccd04ee2a7d","year":2024},{"cited_arxiv_id":"","doi":"","is_internal_anchor":false,"ref_index":2,"title":"Y . Bai, X. Lv, J. Zhang, H. Lyu, J. Tang, Z. Huang, Z. Du, X. Liu, A. Zeng, L. Hou, Y . Dong, J. Tang, and J. Li. Longbench: A bilingual, multitask benchmark for long context understanding, 2023","work_id":"48cbb890-8e16-4eb1-8f1a-de441b4acb71","year":2023},{"cited_arxiv_id":"","doi":"","is_internal_anchor":false,"ref_index":3,"title":"T. B. Brown, B. Mann, N. Ryder, M. Subbiah, J. Kaplan, P. Dhariwal, A. Neelakantan, P. Shyam, G. Sastry, A. Askell, S. Agarwal, A. Herbert-V oss, G. Krueger, T. Henighan, R. Child, A. Ramesh, D. M. Zi","work_id":"46235657-731f-4001-b42a-8f57527cb375","year":2020},{"cited_arxiv_id":"2107.03374","doi":"","is_internal_anchor":true,"ref_index":4,"title":"Evaluating Large Language Models Trained on Code","work_id":"042493e9-b26f-4b4e-bbde-382072ca9b08","year":2021},{"cited_arxiv_id":"2306.15595","doi":"","is_internal_anchor":true,"ref_index":5,"title":"Extending Context Window of Large Language Models via Positional Interpolation","work_id":"c8b6df85-e7da-4bd8-90a4-d309cc2a0f60","year":2023}],"snapshot_sha256":"408a97c7b99f1b8bc4c9ea667a204c525ed44b53e7dff1faae66efb2e5a084ca"},"source":{"id":"2406.12793","kind":"arxiv","version":2},"verdict":{"created_at":"2026-05-11T08:01:06.427306Z","id":"b5a7128a-a712-474f-ab7d-f6dfff956612","model_set":{"reader":"grok-4.3"},"one_line_summary":"GLM-4 models rival or exceed GPT-4 on MMLU, GSM8K, MATH, BBH, GPQA, HumanEval, IFEval, long-context tasks, and Chinese alignment while adding autonomous tool use for web, code, and image generation.","pipeline_version":"pith-pipeline@v0.9.0","pith_extraction_headline":"GLM-4 language models rival or surpass GPT-4 on benchmarks for general ability, reasoning, coding, and Chinese alignment.","strongest_claim":"Evaluations show that GLM-4 1) closely rivals or outperforms GPT-4 in terms of general metrics such as MMLU, GSM8K, MATH, BBH, GPQA, and HumanEval, 2) gets close to GPT-4-Turbo in instruction following as measured by IFEval, 3) matches GPT-4 Turbo (128K) and Claude 3 for long context tasks, and 4) outperforms GPT-4 in Chinese alignments as measured by AlignBench.","weakest_assumption":"That the reported benchmark scores reflect genuine capability rather than test-set contamination, prompt engineering, or selective reporting, given that full training data, exact evaluation protocols, and GLM-4 model weights are not released in the report."}},"verdict_id":"b5a7128a-a712-474f-ab7d-f6dfff956612"}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:db277c3f9737c5034eebde11569e670040bdc2e8f9985edb04bffcf666dac360","target":"record","created_at":"2026-07-05T08:49:53Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"e1d79a50f6830c2440d1a199ae84f45637fb340943fa5dd9637533d3a00c2319","cross_cats_sorted":[],"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-06-18T16:58:21Z","title_canon_sha256":"7e5cbb2258df8a8cea4f770a069588bd123929839363f246b2f828e70f7aed5f"},"schema_version":"1.0","source":{"id":"2406.12793","kind":"arxiv","version":2}},"canonical_sha256":"2c33255a242f7836eb6af34c3891af39b025fa421221e48c93923671fa6c20f9","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"2c33255a242f7836eb6af34c3891af39b025fa421221e48c93923671fa6c20f9","first_computed_at":"2026-07-05T08:49:53.726540Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-07-05T08:49:53.726540Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"8zP5cEdGXj62SFU8q4xtT58yrtPWweIzZ3B3NmrcONzP0f2ryCpkUDLmnbdDs61l36MH2ijtdarQxnt4THt4CQ==","signature_status":"signed_v1","signed_at":"2026-07-05T08:49:53.727038Z","signed_message":"canonical_sha256_bytes"},"source_id":"2406.12793","source_kind":"arxiv","source_version":2}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:db277c3f9737c5034eebde11569e670040bdc2e8f9985edb04bffcf666dac360","sha256:703efb3a46191414549d14f1d76776662271398e9324a616f18710c37bbd3086"],"state_sha256":"42cc3519dac46a91af3eb9f6f9e98952cac1c83fb55b04cd6a7e4c3a017be06c"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"5heGAks8g4PkNDQCYgGd4G98LjV5GvL0iY/nebs9PXRNEt1xMMDDs3Pbox7ectJivG3UkoXulxz9TewB0+XCBw==","signed_message":"bundle_sha256_bytes","signed_at":"2026-08-12T16:42:03.338705Z","bundle_sha256":"cc91ba52c493c8662c2b540a794b259b0a97278b333d6871e66b4a0115f3d0e0"}}