{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:ZY6M52XA4P7AUMSN4XHKQ5DBDO","short_pith_number":"pith:ZY6M52XA","schema_version":"1.0","canonical_sha256":"ce3cceeae0e3fe0a324de5cea874611bbd38785cec7ae02e977ee671aba6e446","source":{"kind":"arxiv","id":"2503.16460","version":1},"attestation_state":"computed","paper":{"title":"Beyond Final Answers: Evaluating Large Language Models for Math Tutoring","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.HC","authors_text":"Adit Gupta, Christopher J. MacLellan, Daniel Weitekamp, Jennifer Reddig, Tommaso Calo","submitted_at":"2025-02-23T15:43:45Z","abstract_excerpt":"Researchers have made notable progress in applying Large Language Models (LLMs) to solve math problems, as demonstrated through efforts like GSM8k, ProofNet, AlphaGeometry, and MathOdyssey. This progress has sparked interest in their potential use for tutoring students in mathematics. However, the reliability of LLMs in tutoring contexts -- where correctness and instructional quality are crucial -- remains underexplored. Moreover, LLM problem-solving capabilities may not necessarily translate into effective tutoring support for students. In this work, we present two novel approaches to evaluat"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2503.16460","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.HC","submitted_at":"2025-02-23T15:43:45Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"d815e66d8c7504c9713f49d6e072a763b34ecb28e497ad5698cd2cb08d2bb229","abstract_canon_sha256":"11afb7e3631f2adfa2962ce28cad2b023b56a0fca6a2798ca688f6901229232d"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:36:41.402230Z","signature_b64":"LpOCfZk7HPJ0MqgzbB1GfEY8ClQgmH/mDKsmK7jIkZsm5V90fHDQeD7cKHYn5KSPr+3T1Ie4tCcfWw2qbZ9TAg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ce3cceeae0e3fe0a324de5cea874611bbd38785cec7ae02e977ee671aba6e446","last_reissued_at":"2026-07-05T10:36:41.401751Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:36:41.401751Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Beyond Final Answers: Evaluating Large Language Models for Math Tutoring","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.HC","authors_text":"Adit Gupta, Christopher J. MacLellan, Daniel Weitekamp, Jennifer Reddig, Tommaso Calo","submitted_at":"2025-02-23T15:43:45Z","abstract_excerpt":"Researchers have made notable progress in applying Large Language Models (LLMs) to solve math problems, as demonstrated through efforts like GSM8k, ProofNet, AlphaGeometry, and MathOdyssey. This progress has sparked interest in their potential use for tutoring students in mathematics. However, the reliability of LLMs in tutoring contexts -- where correctness and instructional quality are crucial -- remains underexplored. Moreover, LLM problem-solving capabilities may not necessarily translate into effective tutoring support for students. In this work, we present two novel approaches to evaluat"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2503.16460","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2503.16460/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2503.16460","created_at":"2026-07-05T10:36:41.401807+00:00"},{"alias_kind":"arxiv_version","alias_value":"2503.16460v1","created_at":"2026-07-05T10:36:41.401807+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2503.16460","created_at":"2026-07-05T10:36:41.401807+00:00"},{"alias_kind":"pith_short_12","alias_value":"ZY6M52XA4P7A","created_at":"2026-07-05T10:36:41.401807+00:00"},{"alias_kind":"pith_short_16","alias_value":"ZY6M52XA4P7AUMSN","created_at":"2026-07-05T10:36:41.401807+00:00"},{"alias_kind":"pith_short_8","alias_value":"ZY6M52XA","created_at":"2026-07-05T10:36:41.401807+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.30774","citing_title":"What Drives Interactive Improvement from Feedback?","ref_index":6,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ZY6M52XA4P7AUMSN4XHKQ5DBDO","json":"https://pith.science/pith/ZY6M52XA4P7AUMSN4XHKQ5DBDO.json","graph_json":"https://pith.science/api/pith-number/ZY6M52XA4P7AUMSN4XHKQ5DBDO/graph.json","events_json":"https://pith.science/api/pith-number/ZY6M52XA4P7AUMSN4XHKQ5DBDO/events.json","paper":"https://pith.science/paper/ZY6M52XA"},"agent_actions":{"view_html":"https://pith.science/pith/ZY6M52XA4P7AUMSN4XHKQ5DBDO","download_json":"https://pith.science/pith/ZY6M52XA4P7AUMSN4XHKQ5DBDO.json","view_paper":"https://pith.science/paper/ZY6M52XA","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2503.16460&json=true","fetch_graph":"https://pith.science/api/pith-number/ZY6M52XA4P7AUMSN4XHKQ5DBDO/graph.json","fetch_events":"https://pith.science/api/pith-number/ZY6M52XA4P7AUMSN4XHKQ5DBDO/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ZY6M52XA4P7AUMSN4XHKQ5DBDO/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ZY6M52XA4P7AUMSN4XHKQ5DBDO/action/storage_attestation","attest_author":"https://pith.science/pith/ZY6M52XA4P7AUMSN4XHKQ5DBDO/action/author_attestation","sign_citation":"https://pith.science/pith/ZY6M52XA4P7AUMSN4XHKQ5DBDO/action/citation_signature","submit_replication":"https://pith.science/pith/ZY6M52XA4P7AUMSN4XHKQ5DBDO/action/replication_record"}},"created_at":"2026-07-05T10:36:41.401807+00:00","updated_at":"2026-07-05T10:36:41.401807+00:00"}