{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:ZM5E6BUMYNEWUJIWWPAZIJSJZO","short_pith_number":"pith:ZM5E6BUM","schema_version":"1.0","canonical_sha256":"cb3a4f068cc3496a2516b3c1942649cba6063f36c780c525e56d3ed495ee8bef","source":{"kind":"arxiv","id":"2410.01748","version":1},"attestation_state":"computed","paper":{"title":"Not All LLM Reasoners Are Created Equal","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Aaron Courville, Alessandro Sordoni, Arian Hosseini, Daniel Toyama, Rishabh Agarwal","submitted_at":"2024-10-02T17:01:10Z","abstract_excerpt":"We study the depth of grade-school math (GSM) problem-solving capabilities of LLMs. To this end, we evaluate their performance on pairs of existing math word problems together so that the answer to the second problem depends on correctly answering the first problem. Our findings reveal a significant reasoning gap in most LLMs, that is performance difference between solving the compositional pairs and solving each question independently. This gap is more pronounced in smaller, more cost-efficient, and math-specialized models. Moreover, instruction-tuning recipes and code generation have varying"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2410.01748","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-sa/4.0/","primary_cat":"cs.LG","submitted_at":"2024-10-02T17:01:10Z","cross_cats_sorted":[],"title_canon_sha256":"2df591f0425718e826a699f24559ea9c364bbd7ab4a60fb7c222003d413658ff","abstract_canon_sha256":"afa6d9cdf884772caa05f62f72b959ccbe74225357f638c4543449e9fe9fb34d"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:14:51.464678Z","signature_b64":"QJxbqbUpkFFZzsewXror6Gu2Mekk0hQfpLB/633s+OD7Lr1GyBy7yOHaY7SEDJvcpO+wjCLYeWW/eRL/9TzTBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"cb3a4f068cc3496a2516b3c1942649cba6063f36c780c525e56d3ed495ee8bef","last_reissued_at":"2026-07-05T09:14:51.464194Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:14:51.464194Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Not All LLM Reasoners Are Created Equal","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Aaron Courville, Alessandro Sordoni, Arian Hosseini, Daniel Toyama, Rishabh Agarwal","submitted_at":"2024-10-02T17:01:10Z","abstract_excerpt":"We study the depth of grade-school math (GSM) problem-solving capabilities of LLMs. To this end, we evaluate their performance on pairs of existing math word problems together so that the answer to the second problem depends on correctly answering the first problem. Our findings reveal a significant reasoning gap in most LLMs, that is performance difference between solving the compositional pairs and solving each question independently. This gap is more pronounced in smaller, more cost-efficient, and math-specialized models. Moreover, instruction-tuning recipes and code generation have varying"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2410.01748","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2410.01748/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2410.01748","created_at":"2026-07-05T09:14:51.464261+00:00"},{"alias_kind":"arxiv_version","alias_value":"2410.01748v1","created_at":"2026-07-05T09:14:51.464261+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2410.01748","created_at":"2026-07-05T09:14:51.464261+00:00"},{"alias_kind":"pith_short_12","alias_value":"ZM5E6BUMYNEW","created_at":"2026-07-05T09:14:51.464261+00:00"},{"alias_kind":"pith_short_16","alias_value":"ZM5E6BUMYNEWUJIW","created_at":"2026-07-05T09:14:51.464261+00:00"},{"alias_kind":"pith_short_8","alias_value":"ZM5E6BUM","created_at":"2026-07-05T09:14:51.464261+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.11470","citing_title":"The Periodic Table of LLM Reasoning: A Structured Survey of Reasoning Paradigms, Methods, and Failure Modes","ref_index":92,"is_internal_anchor":false},{"citing_arxiv_id":"2603.06870","citing_title":"LEAD: Breaking the No-Recovery Bottleneck in Long-Horizon Reasoning","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2603.18203","citing_title":"How Psychological Learning Paradigms Shaped and Constrained Artificial Intelligence","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2501.14249","citing_title":"Humanity's Last Exam","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2604.16646","citing_title":"Agentic Frameworks for Reasoning Tasks: An Empirical Study","ref_index":62,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ZM5E6BUMYNEWUJIWWPAZIJSJZO","json":"https://pith.science/pith/ZM5E6BUMYNEWUJIWWPAZIJSJZO.json","graph_json":"https://pith.science/api/pith-number/ZM5E6BUMYNEWUJIWWPAZIJSJZO/graph.json","events_json":"https://pith.science/api/pith-number/ZM5E6BUMYNEWUJIWWPAZIJSJZO/events.json","paper":"https://pith.science/paper/ZM5E6BUM"},"agent_actions":{"view_html":"https://pith.science/pith/ZM5E6BUMYNEWUJIWWPAZIJSJZO","download_json":"https://pith.science/pith/ZM5E6BUMYNEWUJIWWPAZIJSJZO.json","view_paper":"https://pith.science/paper/ZM5E6BUM","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2410.01748&json=true","fetch_graph":"https://pith.science/api/pith-number/ZM5E6BUMYNEWUJIWWPAZIJSJZO/graph.json","fetch_events":"https://pith.science/api/pith-number/ZM5E6BUMYNEWUJIWWPAZIJSJZO/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ZM5E6BUMYNEWUJIWWPAZIJSJZO/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ZM5E6BUMYNEWUJIWWPAZIJSJZO/action/storage_attestation","attest_author":"https://pith.science/pith/ZM5E6BUMYNEWUJIWWPAZIJSJZO/action/author_attestation","sign_citation":"https://pith.science/pith/ZM5E6BUMYNEWUJIWWPAZIJSJZO/action/citation_signature","submit_replication":"https://pith.science/pith/ZM5E6BUMYNEWUJIWWPAZIJSJZO/action/replication_record"}},"created_at":"2026-07-05T09:14:51.464261+00:00","updated_at":"2026-07-05T09:14:51.464261+00:00"}