{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:KBARRU3CUIHESIDOJFNH6YHTDX","short_pith_number":"pith:KBARRU3C","schema_version":"1.0","canonical_sha256":"504118d362a20e49206e495a7f60f31dd02e9be69476465efe76e89ec89418d0","source":{"kind":"arxiv","id":"2402.19255","version":2},"attestation_state":"computed","paper":{"title":"GSM-Plus: A Comprehensive Benchmark for Evaluating the Robustness of LLMs as Mathematical Problem Solvers","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Leyang Cui, Lingpeng Kong, Qintong Li, Wei Bi, Xueliang Zhao","submitted_at":"2024-02-29T15:26:14Z","abstract_excerpt":"Large language models (LLMs) have achieved impressive performance across various mathematical reasoning benchmarks. However, there are increasing debates regarding whether these models truly understand and apply mathematical knowledge or merely rely on shortcuts for mathematical reasoning. One essential and frequently occurring evidence is that when the math questions are slightly changed, LLMs can behave incorrectly. This motivates us to evaluate the robustness of LLMs' math reasoning capability by testing a wide range of question variations. We introduce the adversarial grade school math (GS"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2402.19255","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-02-29T15:26:14Z","cross_cats_sorted":[],"title_canon_sha256":"d6354b470898a55c4a08ee3bc759e0a8a1444e2703fb5193be0e7cec0232e280","abstract_canon_sha256":"2d9662637a171cbbd7a4cbc3438db773cb887ea649a2c54a55dbe064be8ca415"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:38:57.850984Z","signature_b64":"Q9nb+QvC6ipOngTpciXgC0ThHwewD3M/JitZMLPK5zSd8t8GSKuPCFc/t8m2rmFIb59TUILsvf7fuFURQ9g0CQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"504118d362a20e49206e495a7f60f31dd02e9be69476465efe76e89ec89418d0","last_reissued_at":"2026-07-05T08:38:57.850550Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:38:57.850550Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"GSM-Plus: A Comprehensive Benchmark for Evaluating the Robustness of LLMs as Mathematical Problem Solvers","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Leyang Cui, Lingpeng Kong, Qintong Li, Wei Bi, Xueliang Zhao","submitted_at":"2024-02-29T15:26:14Z","abstract_excerpt":"Large language models (LLMs) have achieved impressive performance across various mathematical reasoning benchmarks. However, there are increasing debates regarding whether these models truly understand and apply mathematical knowledge or merely rely on shortcuts for mathematical reasoning. One essential and frequently occurring evidence is that when the math questions are slightly changed, LLMs can behave incorrectly. This motivates us to evaluate the robustness of LLMs' math reasoning capability by testing a wide range of question variations. We introduce the adversarial grade school math (GS"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2402.19255","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2402.19255/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2402.19255","created_at":"2026-07-05T08:38:57.850607+00:00"},{"alias_kind":"arxiv_version","alias_value":"2402.19255v2","created_at":"2026-07-05T08:38:57.850607+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2402.19255","created_at":"2026-07-05T08:38:57.850607+00:00"},{"alias_kind":"pith_short_12","alias_value":"KBARRU3CUIHE","created_at":"2026-07-05T08:38:57.850607+00:00"},{"alias_kind":"pith_short_16","alias_value":"KBARRU3CUIHESIDO","created_at":"2026-07-05T08:38:57.850607+00:00"},{"alias_kind":"pith_short_8","alias_value":"KBARRU3C","created_at":"2026-07-05T08:38:57.850607+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":9,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.28589","citing_title":"Search for Truth from Reasoning: A Dynamic Representation Editing Framework for Steering LLM Trajectories","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2606.03606","citing_title":"Testing LLM Arithmetic Reasoning Generalization with Automatic Numeric-Remapping Attacks","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2606.28589","citing_title":"Search for Truth from Reasoning: A Dynamic Representation Editing Framework for Steering LLM Trajectories","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2605.31268","citing_title":"Mellum2 Technical Report","ref_index":40,"is_internal_anchor":false},{"citing_arxiv_id":"2507.15698","citing_title":"CoLD: Counterfactually-Guided Length Debiasing for Process Reward Models in Mathematical Reasoning","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2507.04023","citing_title":"Do LLMs Overthink Basic Math Reasoning? Benchmarking the Accuracy-Efficiency Tradeoff in Language Models","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2507.22359","citing_title":"League of LLMs: A Benchmark-Free Paradigm for Mutual Evaluation of Large Language Models","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2503.13657","citing_title":"Why Do Multi-Agent LLM Systems Fail?","ref_index":83,"is_internal_anchor":false},{"citing_arxiv_id":"2604.18738","citing_title":"Remask, Don't Replace: Token-to-Mask Refinement in Diffusion Large Language Models","ref_index":14,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/KBARRU3CUIHESIDOJFNH6YHTDX","json":"https://pith.science/pith/KBARRU3CUIHESIDOJFNH6YHTDX.json","graph_json":"https://pith.science/api/pith-number/KBARRU3CUIHESIDOJFNH6YHTDX/graph.json","events_json":"https://pith.science/api/pith-number/KBARRU3CUIHESIDOJFNH6YHTDX/events.json","paper":"https://pith.science/paper/KBARRU3C"},"agent_actions":{"view_html":"https://pith.science/pith/KBARRU3CUIHESIDOJFNH6YHTDX","download_json":"https://pith.science/pith/KBARRU3CUIHESIDOJFNH6YHTDX.json","view_paper":"https://pith.science/paper/KBARRU3C","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2402.19255&json=true","fetch_graph":"https://pith.science/api/pith-number/KBARRU3CUIHESIDOJFNH6YHTDX/graph.json","fetch_events":"https://pith.science/api/pith-number/KBARRU3CUIHESIDOJFNH6YHTDX/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/KBARRU3CUIHESIDOJFNH6YHTDX/action/timestamp_anchor","attest_storage":"https://pith.science/pith/KBARRU3CUIHESIDOJFNH6YHTDX/action/storage_attestation","attest_author":"https://pith.science/pith/KBARRU3CUIHESIDOJFNH6YHTDX/action/author_attestation","sign_citation":"https://pith.science/pith/KBARRU3CUIHESIDOJFNH6YHTDX/action/citation_signature","submit_replication":"https://pith.science/pith/KBARRU3CUIHESIDOJFNH6YHTDX/action/replication_record"}},"created_at":"2026-07-05T08:38:57.850607+00:00","updated_at":"2026-07-05T08:38:57.850607+00:00"}