{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:DV2QGRNRJWXEKKRFBCYQ2ICE3Q","short_pith_number":"pith:DV2QGRNR","schema_version":"1.0","canonical_sha256":"1d750345b14dae452a2508b10d2044dc3a38a0a892ad8b8ca7b4b431a33e3451","source":{"kind":"arxiv","id":"2409.13729","version":2},"attestation_state":"computed","paper":{"title":"MathGLM-Vision: Solving Mathematical Problems with Multi-Modal Large Language Model","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Bin Xu, Jie Tang, Jinhao Chen, Weihan Wang, Wenmeng Yu, Wenyi Hong, Zhengxiao Du, Zhen Yang, Zhihuan Jiang","submitted_at":"2024-09-10T01:20:22Z","abstract_excerpt":"Large language models (LLMs) have demonstrated significant capabilities in mathematical reasoning, particularly with text-based mathematical problems. However, current multi-modal large language models (MLLMs), especially those specialized in mathematics, tend to focus predominantly on solving geometric problems but ignore the diversity of visual information available in other areas of mathematics. Moreover, the geometric information for these specialized mathematical MLLMs is derived from several public datasets, which are typically limited in diversity and complexity. To address these limita"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2409.13729","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-09-10T01:20:22Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"9564d905b92ca1ef893cdae3defa50a35a39ec14af77f760147afbe74f07a91d","abstract_canon_sha256":"e6d6b8ef3b22cb39a8cb4df677d9aa1c6dd39209740fd2b2cb1281fee08418e3"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:43:09.651169Z","signature_b64":"ntjObsCq6fIIJ56Ondsf5K9Xgsxyd+6Ju2v+Usclbz4bCAHkcLnMT0xioy+TbIUSLhV1XolmsQVCKitmLnJiDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"1d750345b14dae452a2508b10d2044dc3a38a0a892ad8b8ca7b4b431a33e3451","last_reissued_at":"2026-07-05T09:43:09.650582Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:43:09.650582Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"MathGLM-Vision: Solving Mathematical Problems with Multi-Modal Large Language Model","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Bin Xu, Jie Tang, Jinhao Chen, Weihan Wang, Wenmeng Yu, Wenyi Hong, Zhengxiao Du, Zhen Yang, Zhihuan Jiang","submitted_at":"2024-09-10T01:20:22Z","abstract_excerpt":"Large language models (LLMs) have demonstrated significant capabilities in mathematical reasoning, particularly with text-based mathematical problems. However, current multi-modal large language models (MLLMs), especially those specialized in mathematics, tend to focus predominantly on solving geometric problems but ignore the diversity of visual information available in other areas of mathematics. Moreover, the geometric information for these specialized mathematical MLLMs is derived from several public datasets, which are typically limited in diversity and complexity. To address these limita"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2409.13729","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2409.13729/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2409.13729","created_at":"2026-07-05T09:43:09.650648+00:00"},{"alias_kind":"arxiv_version","alias_value":"2409.13729v2","created_at":"2026-07-05T09:43:09.650648+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2409.13729","created_at":"2026-07-05T09:43:09.650648+00:00"},{"alias_kind":"pith_short_12","alias_value":"DV2QGRNRJWXE","created_at":"2026-07-05T09:43:09.650648+00:00"},{"alias_kind":"pith_short_16","alias_value":"DV2QGRNRJWXEKKRF","created_at":"2026-07-05T09:43:09.650648+00:00"},{"alias_kind":"pith_short_8","alias_value":"DV2QGRNR","created_at":"2026-07-05T09:43:09.650648+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2502.02871","citing_title":"Position: Multimodal Large Language Models Can Significantly Advance Scientific Reasoning","ref_index":233,"is_internal_anchor":false},{"citing_arxiv_id":"2503.16549","citing_title":"MathFlow: Enhancing the Perceptual Flow of MLLMs for Visual Mathematical Problems","ref_index":71,"is_internal_anchor":false},{"citing_arxiv_id":"2602.18600","citing_title":"MapTab: A Diagnostic Benchmark for Long-Horizon Multi-Criteria Multimodal Reasoning on Heterogeneous Topological Graphs","ref_index":92,"is_internal_anchor":false},{"citing_arxiv_id":"2411.10442","citing_title":"Enhancing the Reasoning Ability of Multimodal Large Language Models via Mixed Preference Optimization","ref_index":108,"is_internal_anchor":false},{"citing_arxiv_id":"2602.18600","citing_title":"MapTab: A Diagnostic Benchmark for Long-Horizon Multi-Criteria Multimodal Reasoning on Heterogeneous Topological Graphs","ref_index":92,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/DV2QGRNRJWXEKKRFBCYQ2ICE3Q","json":"https://pith.science/pith/DV2QGRNRJWXEKKRFBCYQ2ICE3Q.json","graph_json":"https://pith.science/api/pith-number/DV2QGRNRJWXEKKRFBCYQ2ICE3Q/graph.json","events_json":"https://pith.science/api/pith-number/DV2QGRNRJWXEKKRFBCYQ2ICE3Q/events.json","paper":"https://pith.science/paper/DV2QGRNR"},"agent_actions":{"view_html":"https://pith.science/pith/DV2QGRNRJWXEKKRFBCYQ2ICE3Q","download_json":"https://pith.science/pith/DV2QGRNRJWXEKKRFBCYQ2ICE3Q.json","view_paper":"https://pith.science/paper/DV2QGRNR","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2409.13729&json=true","fetch_graph":"https://pith.science/api/pith-number/DV2QGRNRJWXEKKRFBCYQ2ICE3Q/graph.json","fetch_events":"https://pith.science/api/pith-number/DV2QGRNRJWXEKKRFBCYQ2ICE3Q/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/DV2QGRNRJWXEKKRFBCYQ2ICE3Q/action/timestamp_anchor","attest_storage":"https://pith.science/pith/DV2QGRNRJWXEKKRFBCYQ2ICE3Q/action/storage_attestation","attest_author":"https://pith.science/pith/DV2QGRNRJWXEKKRFBCYQ2ICE3Q/action/author_attestation","sign_citation":"https://pith.science/pith/DV2QGRNRJWXEKKRFBCYQ2ICE3Q/action/citation_signature","submit_replication":"https://pith.science/pith/DV2QGRNRJWXEKKRFBCYQ2ICE3Q/action/replication_record"}},"created_at":"2026-07-05T09:43:09.650648+00:00","updated_at":"2026-07-05T09:43:09.650648+00:00"}