{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:6OFF2RD4G36T7TOB2CCIY5ZAAW","short_pith_number":"pith:6OFF2RD4","schema_version":"1.0","canonical_sha256":"f38a5d447c36fd3fcdc1d0848c772005b6a8437347912b7a3ccfe4d6011651e5","source":{"kind":"arxiv","id":"2407.12023","version":1},"attestation_state":"computed","paper":{"title":"CMMaTH: A Chinese Multi-modal Math Skill Evaluation Benchmark for Foundation Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Cheng-Lin Liu, Fan-Hu Zeng, Fei Yin, Jian Xu, Jia-Xin Zhang, Jin-Feng Bai, Ming-Liang Zhang, Zhen-Ru Pan, Zhi-Long Ji, Zhong-Zhi Li","submitted_at":"2024-06-28T02:35:51Z","abstract_excerpt":"Due to the rapid advancements in multimodal large language models, evaluating their multimodal mathematical capabilities continues to receive wide attention. Despite the datasets like MathVista proposed benchmarks for assessing mathematical capabilities in multimodal scenarios, there is still a lack of corresponding evaluation tools and datasets for fine-grained assessment in the context of K12 education in Chinese language. To systematically evaluate the capability of multimodal large models in solving Chinese multimodal mathematical problems, we propose a Chinese Multi-modal Math Skill Evalu"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2407.12023","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-06-28T02:35:51Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"dba07013c7c78aff4eee30c2a1702225f994d69b3895a671ee3f07ea50e4ec0d","abstract_canon_sha256":"c8dd62289efb96b7d8e97b9827e211aa3b9427711565b48300e8667e1faaef9c"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:44:58.923820Z","signature_b64":"uLmoJN6UkQKE1pTEDoT1Z+Ccv1BFCG1+HQm3YdzkhofarJ7NNnYNokBd14QOoBdj2qRQoiZbrRz3taxGT5erAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f38a5d447c36fd3fcdc1d0848c772005b6a8437347912b7a3ccfe4d6011651e5","last_reissued_at":"2026-07-05T08:44:58.923397Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:44:58.923397Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"CMMaTH: A Chinese Multi-modal Math Skill Evaluation Benchmark for Foundation Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Cheng-Lin Liu, Fan-Hu Zeng, Fei Yin, Jian Xu, Jia-Xin Zhang, Jin-Feng Bai, Ming-Liang Zhang, Zhen-Ru Pan, Zhi-Long Ji, Zhong-Zhi Li","submitted_at":"2024-06-28T02:35:51Z","abstract_excerpt":"Due to the rapid advancements in multimodal large language models, evaluating their multimodal mathematical capabilities continues to receive wide attention. Despite the datasets like MathVista proposed benchmarks for assessing mathematical capabilities in multimodal scenarios, there is still a lack of corresponding evaluation tools and datasets for fine-grained assessment in the context of K12 education in Chinese language. To systematically evaluate the capability of multimodal large models in solving Chinese multimodal mathematical problems, we propose a Chinese Multi-modal Math Skill Evalu"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2407.12023","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2407.12023/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2407.12023","created_at":"2026-07-05T08:44:58.923464+00:00"},{"alias_kind":"arxiv_version","alias_value":"2407.12023v1","created_at":"2026-07-05T08:44:58.923464+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2407.12023","created_at":"2026-07-05T08:44:58.923464+00:00"},{"alias_kind":"pith_short_12","alias_value":"6OFF2RD4G36T","created_at":"2026-07-05T08:44:58.923464+00:00"},{"alias_kind":"pith_short_16","alias_value":"6OFF2RD4G36T7TOB","created_at":"2026-07-05T08:44:58.923464+00:00"},{"alias_kind":"pith_short_8","alias_value":"6OFF2RD4","created_at":"2026-07-05T08:44:58.923464+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2410.04509","citing_title":"ErrorRadar: Benchmarking Complex Mathematical Reasoning of Multimodal Large Language Models Via Error Detection","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2508.06226","citing_title":"GeoLaux: A Benchmark for Evaluating MLLMs' Geometry Performance on Long-Step Problems Requiring Auxiliary Lines","ref_index":18,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/6OFF2RD4G36T7TOB2CCIY5ZAAW","json":"https://pith.science/pith/6OFF2RD4G36T7TOB2CCIY5ZAAW.json","graph_json":"https://pith.science/api/pith-number/6OFF2RD4G36T7TOB2CCIY5ZAAW/graph.json","events_json":"https://pith.science/api/pith-number/6OFF2RD4G36T7TOB2CCIY5ZAAW/events.json","paper":"https://pith.science/paper/6OFF2RD4"},"agent_actions":{"view_html":"https://pith.science/pith/6OFF2RD4G36T7TOB2CCIY5ZAAW","download_json":"https://pith.science/pith/6OFF2RD4G36T7TOB2CCIY5ZAAW.json","view_paper":"https://pith.science/paper/6OFF2RD4","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2407.12023&json=true","fetch_graph":"https://pith.science/api/pith-number/6OFF2RD4G36T7TOB2CCIY5ZAAW/graph.json","fetch_events":"https://pith.science/api/pith-number/6OFF2RD4G36T7TOB2CCIY5ZAAW/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/6OFF2RD4G36T7TOB2CCIY5ZAAW/action/timestamp_anchor","attest_storage":"https://pith.science/pith/6OFF2RD4G36T7TOB2CCIY5ZAAW/action/storage_attestation","attest_author":"https://pith.science/pith/6OFF2RD4G36T7TOB2CCIY5ZAAW/action/author_attestation","sign_citation":"https://pith.science/pith/6OFF2RD4G36T7TOB2CCIY5ZAAW/action/citation_signature","submit_replication":"https://pith.science/pith/6OFF2RD4G36T7TOB2CCIY5ZAAW/action/replication_record"}},"created_at":"2026-07-05T08:44:58.923464+00:00","updated_at":"2026-07-05T08:44:58.923464+00:00"}