{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:2TUQ55VU5H64N3LEITLD7WTD3X","short_pith_number":"pith:2TUQ55VU","schema_version":"1.0","canonical_sha256":"d4e90ef6b4e9fdc6ed6444d63fda63ddc8e63a72bf2d76a2a50b7661d67abaca","source":{"kind":"arxiv","id":"2405.14804","version":4},"attestation_state":"computed","paper":{"title":"Can LLMs Solve longer Math Word Problems Better?","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Can Yang, Tong Xiao, Xin Xu, Yang Wang, Zhenya Huang, Zitong Chao","submitted_at":"2024-05-23T17:13:50Z","abstract_excerpt":"Math Word Problems (MWPs) play a vital role in assessing the capabilities of Large Language Models (LLMs), yet current research primarily focuses on questions with concise contexts. The impact of longer contexts on mathematical reasoning remains under-explored. This study pioneers the investigation of Context Length Generalizability (CoLeG), which refers to the ability of LLMs to solve MWPs with extended narratives. We introduce Extended Grade-School Math (E-GSM), a collection of MWPs featuring lengthy narratives, and propose two novel metrics to evaluate the efficacy and resilience of LLMs in"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2405.14804","kind":"arxiv","version":4},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-05-23T17:13:50Z","cross_cats_sorted":[],"title_canon_sha256":"ec6cc4561005529ed2621adbe862e902bdfcc7d6a6a1d2816ff8ec84f107ab25","abstract_canon_sha256":"522da650a735ff5e1a546dc082975a44cecef6aeb369f14af66b7ca1ffc917d4"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:19:55.239711Z","signature_b64":"+/obBvRLHEEvlP1PFwOls2Wq8ZbQpE8MkhbYLzYfsBxK7lqdQO9vTXynaZC56has0DiLJvzJbRSWhQCBXAJ+CA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"d4e90ef6b4e9fdc6ed6444d63fda63ddc8e63a72bf2d76a2a50b7661d67abaca","last_reissued_at":"2026-07-05T10:19:55.239350Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:19:55.239350Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Can LLMs Solve longer Math Word Problems Better?","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Can Yang, Tong Xiao, Xin Xu, Yang Wang, Zhenya Huang, Zitong Chao","submitted_at":"2024-05-23T17:13:50Z","abstract_excerpt":"Math Word Problems (MWPs) play a vital role in assessing the capabilities of Large Language Models (LLMs), yet current research primarily focuses on questions with concise contexts. The impact of longer contexts on mathematical reasoning remains under-explored. This study pioneers the investigation of Context Length Generalizability (CoLeG), which refers to the ability of LLMs to solve MWPs with extended narratives. We introduce Extended Grade-School Math (E-GSM), a collection of MWPs featuring lengthy narratives, and propose two novel metrics to evaluate the efficacy and resilience of LLMs in"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2405.14804","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2405.14804/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2405.14804","created_at":"2026-07-05T10:19:55.239413+00:00"},{"alias_kind":"arxiv_version","alias_value":"2405.14804v4","created_at":"2026-07-05T10:19:55.239413+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2405.14804","created_at":"2026-07-05T10:19:55.239413+00:00"},{"alias_kind":"pith_short_12","alias_value":"2TUQ55VU5H64","created_at":"2026-07-05T10:19:55.239413+00:00"},{"alias_kind":"pith_short_16","alias_value":"2TUQ55VU5H64N3LE","created_at":"2026-07-05T10:19:55.239413+00:00"},{"alias_kind":"pith_short_8","alias_value":"2TUQ55VU","created_at":"2026-07-05T10:19:55.239413+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/2TUQ55VU5H64N3LEITLD7WTD3X","json":"https://pith.science/pith/2TUQ55VU5H64N3LEITLD7WTD3X.json","graph_json":"https://pith.science/api/pith-number/2TUQ55VU5H64N3LEITLD7WTD3X/graph.json","events_json":"https://pith.science/api/pith-number/2TUQ55VU5H64N3LEITLD7WTD3X/events.json","paper":"https://pith.science/paper/2TUQ55VU"},"agent_actions":{"view_html":"https://pith.science/pith/2TUQ55VU5H64N3LEITLD7WTD3X","download_json":"https://pith.science/pith/2TUQ55VU5H64N3LEITLD7WTD3X.json","view_paper":"https://pith.science/paper/2TUQ55VU","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2405.14804&json=true","fetch_graph":"https://pith.science/api/pith-number/2TUQ55VU5H64N3LEITLD7WTD3X/graph.json","fetch_events":"https://pith.science/api/pith-number/2TUQ55VU5H64N3LEITLD7WTD3X/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/2TUQ55VU5H64N3LEITLD7WTD3X/action/timestamp_anchor","attest_storage":"https://pith.science/pith/2TUQ55VU5H64N3LEITLD7WTD3X/action/storage_attestation","attest_author":"https://pith.science/pith/2TUQ55VU5H64N3LEITLD7WTD3X/action/author_attestation","sign_citation":"https://pith.science/pith/2TUQ55VU5H64N3LEITLD7WTD3X/action/citation_signature","submit_replication":"https://pith.science/pith/2TUQ55VU5H64N3LEITLD7WTD3X/action/replication_record"}},"created_at":"2026-07-05T10:19:55.239413+00:00","updated_at":"2026-07-05T10:19:55.239413+00:00"}