{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:I6PDXMAK4246PTZCMFZ5O36W6J","short_pith_number":"pith:I6PDXMAK","schema_version":"1.0","canonical_sha256":"479e3bb00ae6b9e7cf226173d76fd6f2430eb869a3c38e998bad4a2c5d3175a0","source":{"kind":"arxiv","id":"2402.10104","version":2},"attestation_state":"computed","paper":{"title":"GeoEval: Benchmark for Evaluating LLMs and Multi-Modal Models on Geometry Problem-Solving","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.AI","authors_text":"Chenglin Liu, Fei Yin, Jiaxin Zhang, Mingliang Zhang, Yashar Moshfeghi, Zhongzhi Li","submitted_at":"2024-02-15T16:59:41Z","abstract_excerpt":"Recent advancements in large language models (LLMs) and multi-modal models (MMs) have demonstrated their remarkable capabilities in problem-solving. Yet, their proficiency in tackling geometry math problems, which necessitates an integrated understanding of both textual and visual information, has not been thoroughly evaluated. To address this gap, we introduce the GeoEval benchmark, a comprehensive collection that includes a main subset of 2,000 problems, a 750 problems subset focusing on backward reasoning, an augmented subset of 2,000 problems, and a hard subset of 300 problems. This benchm"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2402.10104","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2024-02-15T16:59:41Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"b289d274189b15e22f9a418643a48850e7f5b20f310021b64f69876508e1a6f0","abstract_canon_sha256":"1eb1ea3f791191b70fdeed6ada95216369b1c0032ec18c6925b83f8b39c131d0"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:20:10.269582Z","signature_b64":"zRzlV/ZRqxnVGBhWSphzxDAp1AS6Dg0ma4O1SLZ+uUghWYSOeoGPhD6uDEViDtceViCvluP2cQ8fhxGKB4FIDA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"479e3bb00ae6b9e7cf226173d76fd6f2430eb869a3c38e998bad4a2c5d3175a0","last_reissued_at":"2026-07-05T08:20:10.269109Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:20:10.269109Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"GeoEval: Benchmark for Evaluating LLMs and Multi-Modal Models on Geometry Problem-Solving","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.AI","authors_text":"Chenglin Liu, Fei Yin, Jiaxin Zhang, Mingliang Zhang, Yashar Moshfeghi, Zhongzhi Li","submitted_at":"2024-02-15T16:59:41Z","abstract_excerpt":"Recent advancements in large language models (LLMs) and multi-modal models (MMs) have demonstrated their remarkable capabilities in problem-solving. Yet, their proficiency in tackling geometry math problems, which necessitates an integrated understanding of both textual and visual information, has not been thoroughly evaluated. To address this gap, we introduce the GeoEval benchmark, a comprehensive collection that includes a main subset of 2,000 problems, a 750 problems subset focusing on backward reasoning, an augmented subset of 2,000 problems, and a hard subset of 300 problems. This benchm"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2402.10104","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2402.10104/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2402.10104","created_at":"2026-07-05T08:20:10.269165+00:00"},{"alias_kind":"arxiv_version","alias_value":"2402.10104v2","created_at":"2026-07-05T08:20:10.269165+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2402.10104","created_at":"2026-07-05T08:20:10.269165+00:00"},{"alias_kind":"pith_short_12","alias_value":"I6PDXMAK4246","created_at":"2026-07-05T08:20:10.269165+00:00"},{"alias_kind":"pith_short_16","alias_value":"I6PDXMAK4246PTZC","created_at":"2026-07-05T08:20:10.269165+00:00"},{"alias_kind":"pith_short_8","alias_value":"I6PDXMAK","created_at":"2026-07-05T08:20:10.269165+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2508.06226","citing_title":"GeoLaux: A Benchmark for Evaluating MLLMs' Geometry Performance on Long-Step Problems Requiring Auxiliary Lines","ref_index":40,"is_internal_anchor":false},{"citing_arxiv_id":"2603.08592","citing_title":"Boosting MLLM Spatial Reasoning with Geometrically Referenced 3D Scene Representations","ref_index":41,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/I6PDXMAK4246PTZCMFZ5O36W6J","json":"https://pith.science/pith/I6PDXMAK4246PTZCMFZ5O36W6J.json","graph_json":"https://pith.science/api/pith-number/I6PDXMAK4246PTZCMFZ5O36W6J/graph.json","events_json":"https://pith.science/api/pith-number/I6PDXMAK4246PTZCMFZ5O36W6J/events.json","paper":"https://pith.science/paper/I6PDXMAK"},"agent_actions":{"view_html":"https://pith.science/pith/I6PDXMAK4246PTZCMFZ5O36W6J","download_json":"https://pith.science/pith/I6PDXMAK4246PTZCMFZ5O36W6J.json","view_paper":"https://pith.science/paper/I6PDXMAK","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2402.10104&json=true","fetch_graph":"https://pith.science/api/pith-number/I6PDXMAK4246PTZCMFZ5O36W6J/graph.json","fetch_events":"https://pith.science/api/pith-number/I6PDXMAK4246PTZCMFZ5O36W6J/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/I6PDXMAK4246PTZCMFZ5O36W6J/action/timestamp_anchor","attest_storage":"https://pith.science/pith/I6PDXMAK4246PTZCMFZ5O36W6J/action/storage_attestation","attest_author":"https://pith.science/pith/I6PDXMAK4246PTZCMFZ5O36W6J/action/author_attestation","sign_citation":"https://pith.science/pith/I6PDXMAK4246PTZCMFZ5O36W6J/action/citation_signature","submit_replication":"https://pith.science/pith/I6PDXMAK4246PTZCMFZ5O36W6J/action/replication_record"}},"created_at":"2026-07-05T08:20:10.269165+00:00","updated_at":"2026-07-05T08:20:10.269165+00:00"}