{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:C76NLIIPS53JNDIWGUQXJ6KQXC","short_pith_number":"pith:C76NLIIP","schema_version":"1.0","canonical_sha256":"17fcd5a10f9776968d16352174f950b899941f872c926ddd7405c349a16c933c","source":{"kind":"arxiv","id":"2312.11370","version":2},"attestation_state":"computed","paper":{"title":"G-LLaVA: Solving Geometric Problem with Multi-Modal Large Language Model","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Hang Xu, Jiacheng Ye, Jiahui Gao, Jianhua Han, Jipeng Zhang, Lanqing Hong, Lingpeng Kong, Renjie Pi, Wanjun Zhong, Yufei Wang, Zhenguo Li","submitted_at":"2023-12-18T17:36:20Z","abstract_excerpt":"Large language models (LLMs) have shown remarkable proficiency in human-level reasoning and generation capabilities, which encourages extensive research on their application in mathematical problem solving. However, current work has been largely focused on text-based mathematical problems, with limited investigation in problems involving geometric information. Addressing this gap, we aim to enable LLMs to solve geometric problems by understanding image input. We first analyze the limitations of current Multimodal Large Language Models (MLLMs) in this area: they struggle to accurately comprehen"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2312.11370","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2023-12-18T17:36:20Z","cross_cats_sorted":[],"title_canon_sha256":"d70f13ca385629d2e13bf22a6cfcc426fde2ae5320904b9cf86e0dd0a932d7f0","abstract_canon_sha256":"d6733f5c9d7409e3fc919c913780f32574886ba8ddbd6a26a657efe4222b02d1"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:56:42.850492Z","signature_b64":"nxd5fEWcSByvn2xKMjsjsUccyo+PTN+ce2UWID0BZHV6cqpH4g9PZnZIgEgENqWIzgGeqJ2gK4fCVuIxpBYQCg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"17fcd5a10f9776968d16352174f950b899941f872c926ddd7405c349a16c933c","last_reissued_at":"2026-07-05T11:56:42.850065Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:56:42.850065Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"G-LLaVA: Solving Geometric Problem with Multi-Modal Large Language Model","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Hang Xu, Jiacheng Ye, Jiahui Gao, Jianhua Han, Jipeng Zhang, Lanqing Hong, Lingpeng Kong, Renjie Pi, Wanjun Zhong, Yufei Wang, Zhenguo Li","submitted_at":"2023-12-18T17:36:20Z","abstract_excerpt":"Large language models (LLMs) have shown remarkable proficiency in human-level reasoning and generation capabilities, which encourages extensive research on their application in mathematical problem solving. However, current work has been largely focused on text-based mathematical problems, with limited investigation in problems involving geometric information. Addressing this gap, we aim to enable LLMs to solve geometric problems by understanding image input. We first analyze the limitations of current Multimodal Large Language Models (MLLMs) in this area: they struggle to accurately comprehen"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2312.11370","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2312.11370/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2312.11370","created_at":"2026-07-05T11:56:42.850123+00:00"},{"alias_kind":"arxiv_version","alias_value":"2312.11370v2","created_at":"2026-07-05T11:56:42.850123+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2312.11370","created_at":"2026-07-05T11:56:42.850123+00:00"},{"alias_kind":"pith_short_12","alias_value":"C76NLIIPS53J","created_at":"2026-07-05T11:56:42.850123+00:00"},{"alias_kind":"pith_short_16","alias_value":"C76NLIIPS53JNDIW","created_at":"2026-07-05T11:56:42.850123+00:00"},{"alias_kind":"pith_short_8","alias_value":"C76NLIIP","created_at":"2026-07-05T11:56:42.850123+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":29,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.08763","citing_title":"OpenCoF: Learning to Reason Through Video Generation","ref_index":7,"is_internal_anchor":true},{"citing_arxiv_id":"2606.24233","citing_title":"Latent Visual States for Efficient Multimodal Reasoning","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2606.18216","citing_title":"Zone of Proximal Policy Optimization: Teacher in Prompts, Not Gradients","ref_index":98,"is_internal_anchor":false},{"citing_arxiv_id":"2606.17888","citing_title":"MathVis-Fine: Aligning Visual Supervision with Necessity via Progressive Dependency-Guided Training for Multimodal Mathematical Reasoning","ref_index":43,"is_internal_anchor":false},{"citing_arxiv_id":"2606.06714","citing_title":"Anchored, Not Graded: Vision-Language Models Fail at Slant-from-Texture Perception","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2606.08728","citing_title":"Artificial Intelligence for Mathematical Reasoning: An Integrated Survey of Language Models, Neuro-symbolic Systems, and Verified Discovery","ref_index":36,"is_internal_anchor":false},{"citing_arxiv_id":"2606.07289","citing_title":"Closed-Form Spectral Regularization for Multi-Task Model Merging","ref_index":79,"is_internal_anchor":false},{"citing_arxiv_id":"2606.06714","citing_title":"Anchored, Not Graded: Vision-Language Models Fail at Slant-from-Texture Perception","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2606.04648","citing_title":"BiNSGPS: Geometry Problem Solving via Bidirectional Neuro-Symbolic Interaction","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2606.01599","citing_title":"TRON: Targeted Rule-Verifiable Online Environments for Visual Reasoning RL","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2606.00390","citing_title":"Zamba2-VL Technical Report","ref_index":135,"is_internal_anchor":false},{"citing_arxiv_id":"2412.04468","citing_title":"NVILA: Efficient Frontier Visual Language Models","ref_index":121,"is_internal_anchor":false},{"citing_arxiv_id":"2503.16549","citing_title":"MathFlow: Enhancing the Perceptual Flow of MLLMs for Visual Mathematical Problems","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2504.09925","citing_title":"FLARE: Fully Integration of Vision-Language Representations for Deep Cross-Modal Understanding","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2605.19852","citing_title":"Are Tools Always Beneficial? Learning to Invoke Tools Adaptively for Dual-Mode Multimodal LLM Reasoning","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2605.20177","citing_title":"From Seeing to Thinking: Decoupling Perception and Reasoning Improves Post-Training of Vision-Language Models","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2506.06856","citing_title":"Vision-EKIPL: External Knowledge-Infused Policy Learning for Visual Reasoning","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2412.14164","citing_title":"MetaMorph: Multimodal Understanding and Generation via Instruction Tuning","ref_index":123,"is_internal_anchor":false},{"citing_arxiv_id":"2403.14624","citing_title":"MathVerse: Does Your Multi-modal LLM Truly See the Diagrams in Visual Math Problems?","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2406.16860","citing_title":"Cambrian-1: A Fully Open, Vision-Centric Exploration of Multimodal LLMs","ref_index":44,"is_internal_anchor":false},{"citing_arxiv_id":"2407.01284","citing_title":"We-Math: Does Your Large Multimodal Model Achieve Human-like Mathematical Reasoning?","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2408.16500","citing_title":"CogVLM2: Visual Language Models for Image and Video Understanding","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2411.10442","citing_title":"Enhancing the Reasoning Ability of Multimodal Large Language Models via Mixed Preference Optimization","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2604.02893","citing_title":"Toward an Artificial General Teacher: Procedural Geometry Data Generation and Visual Grounding with Vision-Language Models","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08560","citing_title":"ZAYA1-VL-8B Technical Report","ref_index":169,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/C76NLIIPS53JNDIWGUQXJ6KQXC","json":"https://pith.science/pith/C76NLIIPS53JNDIWGUQXJ6KQXC.json","graph_json":"https://pith.science/api/pith-number/C76NLIIPS53JNDIWGUQXJ6KQXC/graph.json","events_json":"https://pith.science/api/pith-number/C76NLIIPS53JNDIWGUQXJ6KQXC/events.json","paper":"https://pith.science/paper/C76NLIIP"},"agent_actions":{"view_html":"https://pith.science/pith/C76NLIIPS53JNDIWGUQXJ6KQXC","download_json":"https://pith.science/pith/C76NLIIPS53JNDIWGUQXJ6KQXC.json","view_paper":"https://pith.science/paper/C76NLIIP","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2312.11370&json=true","fetch_graph":"https://pith.science/api/pith-number/C76NLIIPS53JNDIWGUQXJ6KQXC/graph.json","fetch_events":"https://pith.science/api/pith-number/C76NLIIPS53JNDIWGUQXJ6KQXC/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/C76NLIIPS53JNDIWGUQXJ6KQXC/action/timestamp_anchor","attest_storage":"https://pith.science/pith/C76NLIIPS53JNDIWGUQXJ6KQXC/action/storage_attestation","attest_author":"https://pith.science/pith/C76NLIIPS53JNDIWGUQXJ6KQXC/action/author_attestation","sign_citation":"https://pith.science/pith/C76NLIIPS53JNDIWGUQXJ6KQXC/action/citation_signature","submit_replication":"https://pith.science/pith/C76NLIIPS53JNDIWGUQXJ6KQXC/action/replication_record"}},"created_at":"2026-07-05T11:56:42.850123+00:00","updated_at":"2026-07-05T11:56:42.850123+00:00"}