{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:4YN4MVYB3Q2VEM5B7DBG6W2EIY","short_pith_number":"pith:4YN4MVYB","schema_version":"1.0","canonical_sha256":"e61bc65701dc355233a1f8c26f5b444639b562ae342888d37859b060622f7edb","source":{"kind":"arxiv","id":"2407.20311","version":1},"attestation_state":"computed","paper":{"title":"Physics of Language Models: Part 2.1, Grade-School Math and the Hidden Reasoning Process","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL","cs.LG"],"primary_cat":"cs.AI","authors_text":"Tian Ye, Yuanzhi Li, Zeyuan Allen-Zhu, Zicheng Xu","submitted_at":"2024-07-29T17:52:40Z","abstract_excerpt":"Recent advances in language models have demonstrated their capability to solve mathematical reasoning problems, achieving near-perfect accuracy on grade-school level math benchmarks like GSM8K. In this paper, we formally study how language models solve these problems. We design a series of controlled experiments to address several fundamental questions: (1) Can language models truly develop reasoning skills, or do they simply memorize templates? (2) What is the model's hidden (mental) reasoning process? (3) Do models solve math questions using skills similar to or different from humans? (4) Do"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2407.20311","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2024-07-29T17:52:40Z","cross_cats_sorted":["cs.CL","cs.LG"],"title_canon_sha256":"4124aed0d6fd0891fcce317edea3cc9346bdcb69f580dc5e77e55d383a91e756","abstract_canon_sha256":"f35d61c0830eafcb21356a30cc541569ada4bac8ac7f9f3637ab093cb9074185"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:49:56.068163Z","signature_b64":"DNu9lO13NQgDws9I43XH6ZoPXWN6hsKixsamVtCIHi3vI6i8QrshBUj+BJEuUXqSD81psA3uJIX4MXucCR2OBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"e61bc65701dc355233a1f8c26f5b444639b562ae342888d37859b060622f7edb","last_reissued_at":"2026-07-05T08:49:56.067726Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:49:56.067726Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Physics of Language Models: Part 2.1, Grade-School Math and the Hidden Reasoning Process","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL","cs.LG"],"primary_cat":"cs.AI","authors_text":"Tian Ye, Yuanzhi Li, Zeyuan Allen-Zhu, Zicheng Xu","submitted_at":"2024-07-29T17:52:40Z","abstract_excerpt":"Recent advances in language models have demonstrated their capability to solve mathematical reasoning problems, achieving near-perfect accuracy on grade-school level math benchmarks like GSM8K. In this paper, we formally study how language models solve these problems. We design a series of controlled experiments to address several fundamental questions: (1) Can language models truly develop reasoning skills, or do they simply memorize templates? (2) What is the model's hidden (mental) reasoning process? (3) Do models solve math questions using skills similar to or different from humans? (4) Do"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2407.20311","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2407.20311/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2407.20311","created_at":"2026-07-05T08:49:56.067782+00:00"},{"alias_kind":"arxiv_version","alias_value":"2407.20311v1","created_at":"2026-07-05T08:49:56.067782+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2407.20311","created_at":"2026-07-05T08:49:56.067782+00:00"},{"alias_kind":"pith_short_12","alias_value":"4YN4MVYB3Q2V","created_at":"2026-07-05T08:49:56.067782+00:00"},{"alias_kind":"pith_short_16","alias_value":"4YN4MVYB3Q2VEM5B","created_at":"2026-07-05T08:49:56.067782+00:00"},{"alias_kind":"pith_short_8","alias_value":"4YN4MVYB","created_at":"2026-07-05T08:49:56.067782+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":7,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.07678","citing_title":"How Data Shapes RoPE Frequency Usage: From Positional Scale Matching to Length Generalization","ref_index":45,"is_internal_anchor":true},{"citing_arxiv_id":"2412.14164","citing_title":"MetaMorph: Multimodal Understanding and Generation via Instruction Tuning","ref_index":75,"is_internal_anchor":false},{"citing_arxiv_id":"2510.25741","citing_title":"Scaling Latent Reasoning via Looped Language Models","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2410.05229","citing_title":"GSM-Symbolic: Understanding the Limitations of Mathematical Reasoning in Large Language Models","ref_index":100,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12906","citing_title":"Data Difficulty and the Generalization--Extrapolation Tradeoff in LLM Fine-Tuning","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2604.22951","citing_title":"The Power of Power Law: Asymmetry Enables Compositional Reasoning","ref_index":60,"is_internal_anchor":false},{"citing_arxiv_id":"2502.09992","citing_title":"Large Language Diffusion Models","ref_index":35,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/4YN4MVYB3Q2VEM5B7DBG6W2EIY","json":"https://pith.science/pith/4YN4MVYB3Q2VEM5B7DBG6W2EIY.json","graph_json":"https://pith.science/api/pith-number/4YN4MVYB3Q2VEM5B7DBG6W2EIY/graph.json","events_json":"https://pith.science/api/pith-number/4YN4MVYB3Q2VEM5B7DBG6W2EIY/events.json","paper":"https://pith.science/paper/4YN4MVYB"},"agent_actions":{"view_html":"https://pith.science/pith/4YN4MVYB3Q2VEM5B7DBG6W2EIY","download_json":"https://pith.science/pith/4YN4MVYB3Q2VEM5B7DBG6W2EIY.json","view_paper":"https://pith.science/paper/4YN4MVYB","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2407.20311&json=true","fetch_graph":"https://pith.science/api/pith-number/4YN4MVYB3Q2VEM5B7DBG6W2EIY/graph.json","fetch_events":"https://pith.science/api/pith-number/4YN4MVYB3Q2VEM5B7DBG6W2EIY/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/4YN4MVYB3Q2VEM5B7DBG6W2EIY/action/timestamp_anchor","attest_storage":"https://pith.science/pith/4YN4MVYB3Q2VEM5B7DBG6W2EIY/action/storage_attestation","attest_author":"https://pith.science/pith/4YN4MVYB3Q2VEM5B7DBG6W2EIY/action/author_attestation","sign_citation":"https://pith.science/pith/4YN4MVYB3Q2VEM5B7DBG6W2EIY/action/citation_signature","submit_replication":"https://pith.science/pith/4YN4MVYB3Q2VEM5B7DBG6W2EIY/action/replication_record"}},"created_at":"2026-07-05T08:49:56.067782+00:00","updated_at":"2026-07-05T08:49:56.067782+00:00"}