{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:BALQPU6UMT5EKLYFBCJ642BXJ3","short_pith_number":"pith:BALQPU6U","schema_version":"1.0","canonical_sha256":"081707d3d464fa452f050893ee68374ed565cf2894502d00ecabd0f0d4bb1b70","source":{"kind":"arxiv","id":"2504.21233","version":1},"attestation_state":"computed","paper":{"title":"Phi-4-Mini-Reasoning: Exploring the Limits of Small Reasoning Language Models in Math","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Baolin Peng, Dongdong Chen, Hany Awadalla, Haoran Xu, Jianfeng Gao, Liliang Ren, Mei Gao, Shuohang Wang, Weijian Xu, Weizhu Chen, Yelong Shen, Yen-Chun Chen, Young Jin Kim, Yunsheng Li","submitted_at":"2025-04-30T00:04:35Z","abstract_excerpt":"Chain-of-Thought (CoT) significantly enhances formal reasoning capabilities in Large Language Models (LLMs) by training them to explicitly generate intermediate reasoning steps. While LLMs readily benefit from such techniques, improving reasoning in Small Language Models (SLMs) remains challenging due to their limited model capacity. Recent work by Deepseek-R1 demonstrates that distillation from LLM-generated synthetic data can substantially improve the reasoning ability of SLM. However, the detailed modeling recipe is not disclosed. In this work, we present a systematic training recipe for SL"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2504.21233","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2025-04-30T00:04:35Z","cross_cats_sorted":[],"title_canon_sha256":"16aea4d1ae2281958c4da815dd2b6679fcd37d043f6a94a95a4e52dfb886246b","abstract_canon_sha256":"46629648776789e5dad2caa90d07846e84e848bb592bc9a8c5a3643c3a797156"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:56:20.433206Z","signature_b64":"D9sTmiyGFudNAbzEiWvWCJPTkQ3Nc75/+cjVRG1oyFG5CIcoivjCX2abRjDur0irasIMPwFvSJg/a4yy80ozCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"081707d3d464fa452f050893ee68374ed565cf2894502d00ecabd0f0d4bb1b70","last_reissued_at":"2026-07-05T10:56:20.432718Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:56:20.432718Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Phi-4-Mini-Reasoning: Exploring the Limits of Small Reasoning Language Models in Math","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Baolin Peng, Dongdong Chen, Hany Awadalla, Haoran Xu, Jianfeng Gao, Liliang Ren, Mei Gao, Shuohang Wang, Weijian Xu, Weizhu Chen, Yelong Shen, Yen-Chun Chen, Young Jin Kim, Yunsheng Li","submitted_at":"2025-04-30T00:04:35Z","abstract_excerpt":"Chain-of-Thought (CoT) significantly enhances formal reasoning capabilities in Large Language Models (LLMs) by training them to explicitly generate intermediate reasoning steps. While LLMs readily benefit from such techniques, improving reasoning in Small Language Models (SLMs) remains challenging due to their limited model capacity. Recent work by Deepseek-R1 demonstrates that distillation from LLM-generated synthetic data can substantially improve the reasoning ability of SLM. However, the detailed modeling recipe is not disclosed. In this work, we present a systematic training recipe for SL"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2504.21233","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2504.21233/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2504.21233","created_at":"2026-07-05T10:56:20.432778+00:00"},{"alias_kind":"arxiv_version","alias_value":"2504.21233v1","created_at":"2026-07-05T10:56:20.432778+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2504.21233","created_at":"2026-07-05T10:56:20.432778+00:00"},{"alias_kind":"pith_short_12","alias_value":"BALQPU6UMT5E","created_at":"2026-07-05T10:56:20.432778+00:00"},{"alias_kind":"pith_short_16","alias_value":"BALQPU6UMT5EKLYF","created_at":"2026-07-05T10:56:20.432778+00:00"},{"alias_kind":"pith_short_8","alias_value":"BALQPU6U","created_at":"2026-07-05T10:56:20.432778+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.06021","citing_title":"OPRD: On-Policy Representation Distillation","ref_index":38,"is_internal_anchor":false},{"citing_arxiv_id":"2605.29247","citing_title":"DenseSteer: Steering Small Language Models towards Dense Math Reasoning","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2509.20241","citing_title":"Energy Use of AI Inference, Efficiency Pathways, and Test-Time Scaling","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2604.12229","citing_title":"HintMR: Eliciting Stronger Mathematical Reasoning in Small Language Models","ref_index":4,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/BALQPU6UMT5EKLYFBCJ642BXJ3","json":"https://pith.science/pith/BALQPU6UMT5EKLYFBCJ642BXJ3.json","graph_json":"https://pith.science/api/pith-number/BALQPU6UMT5EKLYFBCJ642BXJ3/graph.json","events_json":"https://pith.science/api/pith-number/BALQPU6UMT5EKLYFBCJ642BXJ3/events.json","paper":"https://pith.science/paper/BALQPU6U"},"agent_actions":{"view_html":"https://pith.science/pith/BALQPU6UMT5EKLYFBCJ642BXJ3","download_json":"https://pith.science/pith/BALQPU6UMT5EKLYFBCJ642BXJ3.json","view_paper":"https://pith.science/paper/BALQPU6U","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2504.21233&json=true","fetch_graph":"https://pith.science/api/pith-number/BALQPU6UMT5EKLYFBCJ642BXJ3/graph.json","fetch_events":"https://pith.science/api/pith-number/BALQPU6UMT5EKLYFBCJ642BXJ3/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/BALQPU6UMT5EKLYFBCJ642BXJ3/action/timestamp_anchor","attest_storage":"https://pith.science/pith/BALQPU6UMT5EKLYFBCJ642BXJ3/action/storage_attestation","attest_author":"https://pith.science/pith/BALQPU6UMT5EKLYFBCJ642BXJ3/action/author_attestation","sign_citation":"https://pith.science/pith/BALQPU6UMT5EKLYFBCJ642BXJ3/action/citation_signature","submit_replication":"https://pith.science/pith/BALQPU6UMT5EKLYFBCJ642BXJ3/action/replication_record"}},"created_at":"2026-07-05T10:56:20.432778+00:00","updated_at":"2026-07-05T10:56:20.432778+00:00"}