{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:3FG6JS7NHQHPMFVXBP3BKYAAAQ","short_pith_number":"pith:3FG6JS7N","schema_version":"1.0","canonical_sha256":"d94de4cbed3c0ef616b70bf6156000041be4bebf51fe84b79c81e7e3d024310e","source":{"kind":"arxiv","id":"2503.10573","version":2},"attestation_state":"computed","paper":{"title":"Evaluating Mathematical Reasoning Across Large Language Models: A Fine-Grained Approach","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Afrar Jahin, Arif Hassan Zidan, Tianming Liu, Wei Zhang, Yu Bao","submitted_at":"2025-03-13T17:23:45Z","abstract_excerpt":"With the rapid advancement of Artificial Intelligence (AI), Large Language Models (LLMs) have significantly impacted a wide array of domains, including healthcare, engineering, science, education, and mathematical reasoning. Among these, mathematical reasoning remains a particularly challenging capability, often requiring multi-step logic and abstract generalization. While prior work has explored LLM performance on reasoning tasks, comprehensive evaluations that span both depth and breadth across model families remain limited. In this study, we present a systematic evaluation of mathematical r"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2503.10573","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2025-03-13T17:23:45Z","cross_cats_sorted":[],"title_canon_sha256":"b87ddfddae1ca8728521ecea93b2d81b5d6a2a9d6880b4eed20d78904f7dea0c","abstract_canon_sha256":"57b664ed3bc72e40237efee6e32419b0777ab7dcfb9a11421c5f02103e98fdb5"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:05:23.818491Z","signature_b64":"uAgW54yvxlcbLaNJDODUYhGMjrxYm8ORv3QlBIYAF780w3zAyTbHHx1jknnRXKeMDQOkWg7UnPMBP0LVNU3OCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"d94de4cbed3c0ef616b70bf6156000041be4bebf51fe84b79c81e7e3d024310e","last_reissued_at":"2026-07-05T11:05:23.817991Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:05:23.817991Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Evaluating Mathematical Reasoning Across Large Language Models: A Fine-Grained Approach","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Afrar Jahin, Arif Hassan Zidan, Tianming Liu, Wei Zhang, Yu Bao","submitted_at":"2025-03-13T17:23:45Z","abstract_excerpt":"With the rapid advancement of Artificial Intelligence (AI), Large Language Models (LLMs) have significantly impacted a wide array of domains, including healthcare, engineering, science, education, and mathematical reasoning. Among these, mathematical reasoning remains a particularly challenging capability, often requiring multi-step logic and abstract generalization. While prior work has explored LLM performance on reasoning tasks, comprehensive evaluations that span both depth and breadth across model families remain limited. In this study, we present a systematic evaluation of mathematical r"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2503.10573","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2503.10573/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2503.10573","created_at":"2026-07-05T11:05:23.818056+00:00"},{"alias_kind":"arxiv_version","alias_value":"2503.10573v2","created_at":"2026-07-05T11:05:23.818056+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2503.10573","created_at":"2026-07-05T11:05:23.818056+00:00"},{"alias_kind":"pith_short_12","alias_value":"3FG6JS7NHQHP","created_at":"2026-07-05T11:05:23.818056+00:00"},{"alias_kind":"pith_short_16","alias_value":"3FG6JS7NHQHPMFVX","created_at":"2026-07-05T11:05:23.818056+00:00"},{"alias_kind":"pith_short_8","alias_value":"3FG6JS7N","created_at":"2026-07-05T11:05:23.818056+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/3FG6JS7NHQHPMFVXBP3BKYAAAQ","json":"https://pith.science/pith/3FG6JS7NHQHPMFVXBP3BKYAAAQ.json","graph_json":"https://pith.science/api/pith-number/3FG6JS7NHQHPMFVXBP3BKYAAAQ/graph.json","events_json":"https://pith.science/api/pith-number/3FG6JS7NHQHPMFVXBP3BKYAAAQ/events.json","paper":"https://pith.science/paper/3FG6JS7N"},"agent_actions":{"view_html":"https://pith.science/pith/3FG6JS7NHQHPMFVXBP3BKYAAAQ","download_json":"https://pith.science/pith/3FG6JS7NHQHPMFVXBP3BKYAAAQ.json","view_paper":"https://pith.science/paper/3FG6JS7N","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2503.10573&json=true","fetch_graph":"https://pith.science/api/pith-number/3FG6JS7NHQHPMFVXBP3BKYAAAQ/graph.json","fetch_events":"https://pith.science/api/pith-number/3FG6JS7NHQHPMFVXBP3BKYAAAQ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/3FG6JS7NHQHPMFVXBP3BKYAAAQ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/3FG6JS7NHQHPMFVXBP3BKYAAAQ/action/storage_attestation","attest_author":"https://pith.science/pith/3FG6JS7NHQHPMFVXBP3BKYAAAQ/action/author_attestation","sign_citation":"https://pith.science/pith/3FG6JS7NHQHPMFVXBP3BKYAAAQ/action/citation_signature","submit_replication":"https://pith.science/pith/3FG6JS7NHQHPMFVXBP3BKYAAAQ/action/replication_record"}},"created_at":"2026-07-05T11:05:23.818056+00:00","updated_at":"2026-07-05T11:05:23.818056+00:00"}