{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:65S5CVSV3WONUSSAPEMKYIFZKO","short_pith_number":"pith:65S5CVSV","schema_version":"1.0","canonical_sha256":"f765d15655dd9cda4a407918ac20b9538f8f92ff5db6b41a8558eafabb58dd3b","source":{"kind":"arxiv","id":"2507.22958","version":1},"attestation_state":"computed","paper":{"title":"CHECK-MAT: Checking Hand-Written Mathematical Answers for the Russian Unified State Exam","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CV","authors_text":"Ruslan Khrulev","submitted_at":"2025-07-29T23:46:45Z","abstract_excerpt":"This paper introduces a novel benchmark, EGE-Math Solutions Assessment Benchmark, for evaluating Vision-Language Models (VLMs) on their ability to assess hand-written mathematical solutions. Unlike existing benchmarks that focus on problem solving, our approach centres on understanding student solutions, identifying mistakes, and assigning grades according to fixed criteria. We compile 122 scanned solutions from the Russian Unified State Exam (EGE) together with official expert grades, and evaluate seven modern VLMs from Google, OpenAI, Arcee AI, and Alibaba Cloud in three inference modes. The"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2507.22958","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CV","submitted_at":"2025-07-29T23:46:45Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"29bce09e4ae777cec1fa70567ad64777859ba0a62203afa333f530fe9e04c7f1","abstract_canon_sha256":"208d320184549cfd162de2f76efdfffb045aab36025592851e2822257dc058d4"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:46:02.858491Z","signature_b64":"Jjc09fIx9E40b2iDxgMtVNI9eS+M2lrPRUa9fONiI+7c81g5y4prgO21KNroezORDLuzOQwv4WFjbgQo6B2ZAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f765d15655dd9cda4a407918ac20b9538f8f92ff5db6b41a8558eafabb58dd3b","last_reissued_at":"2026-07-05T11:46:02.857844Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:46:02.857844Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"CHECK-MAT: Checking Hand-Written Mathematical Answers for the Russian Unified State Exam","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CV","authors_text":"Ruslan Khrulev","submitted_at":"2025-07-29T23:46:45Z","abstract_excerpt":"This paper introduces a novel benchmark, EGE-Math Solutions Assessment Benchmark, for evaluating Vision-Language Models (VLMs) on their ability to assess hand-written mathematical solutions. Unlike existing benchmarks that focus on problem solving, our approach centres on understanding student solutions, identifying mistakes, and assigning grades according to fixed criteria. We compile 122 scanned solutions from the Russian Unified State Exam (EGE) together with official expert grades, and evaluate seven modern VLMs from Google, OpenAI, Arcee AI, and Alibaba Cloud in three inference modes. The"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2507.22958","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2507.22958/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2507.22958","created_at":"2026-07-05T11:46:02.857898+00:00"},{"alias_kind":"arxiv_version","alias_value":"2507.22958v1","created_at":"2026-07-05T11:46:02.857898+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2507.22958","created_at":"2026-07-05T11:46:02.857898+00:00"},{"alias_kind":"pith_short_12","alias_value":"65S5CVSV3WON","created_at":"2026-07-05T11:46:02.857898+00:00"},{"alias_kind":"pith_short_16","alias_value":"65S5CVSV3WONUSSA","created_at":"2026-07-05T11:46:02.857898+00:00"},{"alias_kind":"pith_short_8","alias_value":"65S5CVSV","created_at":"2026-07-05T11:46:02.857898+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2602.00095","citing_title":"EDU-CIRCUIT-HW: Evaluating Multimodal Large Language Models on Real-World University-Level STEM Student Handwritten Solutions","ref_index":1,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/65S5CVSV3WONUSSAPEMKYIFZKO","json":"https://pith.science/pith/65S5CVSV3WONUSSAPEMKYIFZKO.json","graph_json":"https://pith.science/api/pith-number/65S5CVSV3WONUSSAPEMKYIFZKO/graph.json","events_json":"https://pith.science/api/pith-number/65S5CVSV3WONUSSAPEMKYIFZKO/events.json","paper":"https://pith.science/paper/65S5CVSV"},"agent_actions":{"view_html":"https://pith.science/pith/65S5CVSV3WONUSSAPEMKYIFZKO","download_json":"https://pith.science/pith/65S5CVSV3WONUSSAPEMKYIFZKO.json","view_paper":"https://pith.science/paper/65S5CVSV","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2507.22958&json=true","fetch_graph":"https://pith.science/api/pith-number/65S5CVSV3WONUSSAPEMKYIFZKO/graph.json","fetch_events":"https://pith.science/api/pith-number/65S5CVSV3WONUSSAPEMKYIFZKO/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/65S5CVSV3WONUSSAPEMKYIFZKO/action/timestamp_anchor","attest_storage":"https://pith.science/pith/65S5CVSV3WONUSSAPEMKYIFZKO/action/storage_attestation","attest_author":"https://pith.science/pith/65S5CVSV3WONUSSAPEMKYIFZKO/action/author_attestation","sign_citation":"https://pith.science/pith/65S5CVSV3WONUSSAPEMKYIFZKO/action/citation_signature","submit_replication":"https://pith.science/pith/65S5CVSV3WONUSSAPEMKYIFZKO/action/replication_record"}},"created_at":"2026-07-05T11:46:02.857898+00:00","updated_at":"2026-07-05T11:46:02.857898+00:00"}