{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:PJ7KVVZ2NERH5NYOKMFGFHMGXC","short_pith_number":"pith:PJ7KVVZ2","schema_version":"1.0","canonical_sha256":"7a7eaad73a69227eb70e530a629d86b89506a6a184645a39f4630a8dc0d2a54e","source":{"kind":"arxiv","id":"2402.00157","version":4},"attestation_state":"computed","paper":{"title":"Large Language Models for Mathematical Reasoning: Progresses and Challenges","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Di Liu, Janice Ahn, Renze Lou, Rishu Verma, Rui Zhang, Wenpeng Yin","submitted_at":"2024-01-31T20:26:32Z","abstract_excerpt":"Mathematical reasoning serves as a cornerstone for assessing the fundamental cognitive capabilities of human intelligence. In recent times, there has been a notable surge in the development of Large Language Models (LLMs) geared towards the automated resolution of mathematical problems. However, the landscape of mathematical problem types is vast and varied, with LLM-oriented techniques undergoing evaluation across diverse datasets and settings. This diversity makes it challenging to discern the true advancements and obstacles within this burgeoning field. This survey endeavors to address four"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2402.00157","kind":"arxiv","version":4},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-01-31T20:26:32Z","cross_cats_sorted":[],"title_canon_sha256":"5a19becaacac49ad000ad4775e5fc527546493792cbdfd5409439d4c47ee009e","abstract_canon_sha256":"f64ed7e8cf513f2763bac31989667cbf9528fb571826975f52b25c639e379ea8"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:07:45.934118Z","signature_b64":"m0GqMnWemFAmTyW+erW7BfZ8OksD6rJQbiyUWMwBswv/u268ktoCoFzGFgtplItu5IeAFdOxCUmZZ8gqOdEuBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"7a7eaad73a69227eb70e530a629d86b89506a6a184645a39f4630a8dc0d2a54e","last_reissued_at":"2026-07-05T09:07:45.933619Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:07:45.933619Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Large Language Models for Mathematical Reasoning: Progresses and Challenges","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Di Liu, Janice Ahn, Renze Lou, Rishu Verma, Rui Zhang, Wenpeng Yin","submitted_at":"2024-01-31T20:26:32Z","abstract_excerpt":"Mathematical reasoning serves as a cornerstone for assessing the fundamental cognitive capabilities of human intelligence. In recent times, there has been a notable surge in the development of Large Language Models (LLMs) geared towards the automated resolution of mathematical problems. However, the landscape of mathematical problem types is vast and varied, with LLM-oriented techniques undergoing evaluation across diverse datasets and settings. This diversity makes it challenging to discern the true advancements and obstacles within this burgeoning field. This survey endeavors to address four"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2402.00157","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2402.00157/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2402.00157","created_at":"2026-07-05T09:07:45.933676+00:00"},{"alias_kind":"arxiv_version","alias_value":"2402.00157v4","created_at":"2026-07-05T09:07:45.933676+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2402.00157","created_at":"2026-07-05T09:07:45.933676+00:00"},{"alias_kind":"pith_short_12","alias_value":"PJ7KVVZ2NERH","created_at":"2026-07-05T09:07:45.933676+00:00"},{"alias_kind":"pith_short_16","alias_value":"PJ7KVVZ2NERH5NYO","created_at":"2026-07-05T09:07:45.933676+00:00"},{"alias_kind":"pith_short_8","alias_value":"PJ7KVVZ2","created_at":"2026-07-05T09:07:45.933676+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":25,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.23791","citing_title":"One Generator, Any Process: LLM-Conditioning for the LHC","ref_index":244,"is_internal_anchor":false},{"citing_arxiv_id":"2606.18453","citing_title":"LLM Parameters for Math Across Languages: Shared or Separate?","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2607.01678","citing_title":"SCAPE: Accurate and Efficient LLM Training with Extreme Sparse Communication","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2607.00147","citing_title":"RareDxR1: Autonomous Medical Reasoning for Rare Disease Diagnosis Beyond Human Annotation","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2606.23791","citing_title":"One Generator, Any Process: LLM-Conditioning for the LHC","ref_index":248,"is_internal_anchor":false},{"citing_arxiv_id":"2605.29822","citing_title":"Inferring Code Correctness from Specification","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2410.04509","citing_title":"ErrorRadar: Benchmarking Complex Mathematical Reasoning of Multimodal Large Language Models Via Error Detection","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2410.04047","citing_title":"TS-Reasoner: Domain-Oriented Time Series Inference Agents for Reasoning and Automated Analysis","ref_index":38,"is_internal_anchor":false},{"citing_arxiv_id":"2503.22693","citing_title":"Bridging Language Models and Financial Analysis","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2503.16549","citing_title":"MathFlow: Enhancing the Perceptual Flow of MLLMs for Visual Mathematical Problems","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2504.09775","citing_title":"MIST: A Co-Design Framework for Heterogeneous, Multi-Stage LLM Inference","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2605.22211","citing_title":"CLORE: Content-Level Optimization for Reasoning Efficiency","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2507.10614","citing_title":"Fine-tuning Large Language Model for Automated Algorithm Design","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2510.16079","citing_title":"EvolveR: Self-Evolving LLM Agents through an Experience-Driven Lifecycle","ref_index":40,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10923","citing_title":"Dynamic Skill Lifecycle Management for Agentic Reinforcement Learning","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2505.19237","citing_title":"Sensorimotor Self-Recognition in Multimodal Large Language Model-Driven Robots","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2506.01770","citing_title":"ReGA: Model-Based Safeguard for LLMs via Representation-Guided Abstraction","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2507.22359","citing_title":"League of LLMs: A Benchmark-Free Paradigm for Mutual Evaluation of Large Language Models","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2510.15494","citing_title":"Do AI Models Dream of Faster Code? An Empirical Study on LLM-Proposed Performance Improvements in Real-World Software","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2510.16079","citing_title":"EvolveR: Self-Evolving LLM Agents through an Experience-Driven Lifecycle","ref_index":40,"is_internal_anchor":false},{"citing_arxiv_id":"2602.02280","citing_title":"RACC: Representation-Aware Coverage Criteria for LLM Safety Testing","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2503.10615","citing_title":"R1-Onevision: Advancing Generalized Multimodal Reasoning through Cross-Modal Formalization","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2605.10923","citing_title":"Dynamic Skill Lifecycle Management for Agentic Reinforcement Learning","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2604.23712","citing_title":"OptProver: Bridging Olympiad and Optimization through Continual Training in Formal Theorem Proving","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2605.04064","citing_title":"Improving Medical VQA through Trajectory-Aware Process Supervision","ref_index":1,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/PJ7KVVZ2NERH5NYOKMFGFHMGXC","json":"https://pith.science/pith/PJ7KVVZ2NERH5NYOKMFGFHMGXC.json","graph_json":"https://pith.science/api/pith-number/PJ7KVVZ2NERH5NYOKMFGFHMGXC/graph.json","events_json":"https://pith.science/api/pith-number/PJ7KVVZ2NERH5NYOKMFGFHMGXC/events.json","paper":"https://pith.science/paper/PJ7KVVZ2"},"agent_actions":{"view_html":"https://pith.science/pith/PJ7KVVZ2NERH5NYOKMFGFHMGXC","download_json":"https://pith.science/pith/PJ7KVVZ2NERH5NYOKMFGFHMGXC.json","view_paper":"https://pith.science/paper/PJ7KVVZ2","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2402.00157&json=true","fetch_graph":"https://pith.science/api/pith-number/PJ7KVVZ2NERH5NYOKMFGFHMGXC/graph.json","fetch_events":"https://pith.science/api/pith-number/PJ7KVVZ2NERH5NYOKMFGFHMGXC/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/PJ7KVVZ2NERH5NYOKMFGFHMGXC/action/timestamp_anchor","attest_storage":"https://pith.science/pith/PJ7KVVZ2NERH5NYOKMFGFHMGXC/action/storage_attestation","attest_author":"https://pith.science/pith/PJ7KVVZ2NERH5NYOKMFGFHMGXC/action/author_attestation","sign_citation":"https://pith.science/pith/PJ7KVVZ2NERH5NYOKMFGFHMGXC/action/citation_signature","submit_replication":"https://pith.science/pith/PJ7KVVZ2NERH5NYOKMFGFHMGXC/action/replication_record"}},"created_at":"2026-07-05T09:07:45.933676+00:00","updated_at":"2026-07-05T09:07:45.933676+00:00"}