{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:7WHYFJWJBUQTNSFRM4WHRHFB3S","short_pith_number":"pith:7WHYFJWJ","schema_version":"1.0","canonical_sha256":"fd8f82a6c90d2136c8b1672c789ca1dc95eca730318912622db9a9e1860367b0","source":{"kind":"arxiv","id":"2502.18001","version":3},"attestation_state":"computed","paper":{"title":"Unveiling the Key Factors for Distilling Chain-of-Thought Reasoning","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Dietrich Klakow, Hui Su, Miaoran Zhang, Wenjie Li, Wenjin Guo, Xiaoyu Shen, Xinghao Chen, Yanjun Chen, Yijie Pan, Yirong Sun, Zhijing Sun","submitted_at":"2025-02-25T09:08:45Z","abstract_excerpt":"Large Language Models (LLMs) excel in reasoning tasks through Chain-of-Thought (CoT) prompting. However, CoT prompting greatly increases computational demands, which has prompted growing interest in distilling CoT capabilities into Small Language Models (SLMs). This study systematically examines the factors influencing CoT distillation, including the choice of granularity, format and teacher model. Through experiments involving four teacher models and seven student models across seven mathematical and commonsense reasoning datasets, we uncover three key findings: (1) Unlike LLMs, SLMs exhibit "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.18001","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.CL","submitted_at":"2025-02-25T09:08:45Z","cross_cats_sorted":[],"title_canon_sha256":"1ddb90894843ebf0f7bb8287eb05fa8d68a50fb13e85c0ff05bd45498066269c","abstract_canon_sha256":"966d7cc953babed7f609f54c35e5011d1898a999ffa4abf91092667ee55fa0a1"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:10:05.747556Z","signature_b64":"003sLXuBhVUpmogAiTqGjseBCYHmzaJLQfemowAJbkvRNBu5OjvtLRtD0KV3zXkUGVf0c3wOjMGvLXfJyf/uBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"fd8f82a6c90d2136c8b1672c789ca1dc95eca730318912622db9a9e1860367b0","last_reissued_at":"2026-07-05T11:10:05.747046Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:10:05.747046Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Unveiling the Key Factors for Distilling Chain-of-Thought Reasoning","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Dietrich Klakow, Hui Su, Miaoran Zhang, Wenjie Li, Wenjin Guo, Xiaoyu Shen, Xinghao Chen, Yanjun Chen, Yijie Pan, Yirong Sun, Zhijing Sun","submitted_at":"2025-02-25T09:08:45Z","abstract_excerpt":"Large Language Models (LLMs) excel in reasoning tasks through Chain-of-Thought (CoT) prompting. However, CoT prompting greatly increases computational demands, which has prompted growing interest in distilling CoT capabilities into Small Language Models (SLMs). This study systematically examines the factors influencing CoT distillation, including the choice of granularity, format and teacher model. Through experiments involving four teacher models and seven student models across seven mathematical and commonsense reasoning datasets, we uncover three key findings: (1) Unlike LLMs, SLMs exhibit "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.18001","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.18001/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.18001","created_at":"2026-07-05T11:10:05.747109+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.18001v3","created_at":"2026-07-05T11:10:05.747109+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.18001","created_at":"2026-07-05T11:10:05.747109+00:00"},{"alias_kind":"pith_short_12","alias_value":"7WHYFJWJBUQT","created_at":"2026-07-05T11:10:05.747109+00:00"},{"alias_kind":"pith_short_16","alias_value":"7WHYFJWJBUQTNSFR","created_at":"2026-07-05T11:10:05.747109+00:00"},{"alias_kind":"pith_short_8","alias_value":"7WHYFJWJ","created_at":"2026-07-05T11:10:05.747109+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.08268","citing_title":"Different Teachers, Different Capabilities: Sub-1B On-Device Distillation for Structured Text Enrichment","ref_index":13,"is_internal_anchor":true},{"citing_arxiv_id":"2606.06840","citing_title":"Characterize Then Distill: Mechanistic Reasoning in Large Output Spaces","ref_index":159,"is_internal_anchor":false},{"citing_arxiv_id":"2601.13992","citing_title":"\"The Whole Is Greater Than the Sum of Its Parts\": A Compatibility-Aware Multi-Teacher CoT Distillation Framework","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2503.09567","citing_title":"Towards Reasoning Era: A Survey of Long Chain-of-Thought for Reasoning Large Language Models","ref_index":105,"is_internal_anchor":false},{"citing_arxiv_id":"2604.18936","citing_title":"Fine-Tuning Small Reasoning Models for Quantum Field Theory","ref_index":43,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/7WHYFJWJBUQTNSFRM4WHRHFB3S","json":"https://pith.science/pith/7WHYFJWJBUQTNSFRM4WHRHFB3S.json","graph_json":"https://pith.science/api/pith-number/7WHYFJWJBUQTNSFRM4WHRHFB3S/graph.json","events_json":"https://pith.science/api/pith-number/7WHYFJWJBUQTNSFRM4WHRHFB3S/events.json","paper":"https://pith.science/paper/7WHYFJWJ"},"agent_actions":{"view_html":"https://pith.science/pith/7WHYFJWJBUQTNSFRM4WHRHFB3S","download_json":"https://pith.science/pith/7WHYFJWJBUQTNSFRM4WHRHFB3S.json","view_paper":"https://pith.science/paper/7WHYFJWJ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.18001&json=true","fetch_graph":"https://pith.science/api/pith-number/7WHYFJWJBUQTNSFRM4WHRHFB3S/graph.json","fetch_events":"https://pith.science/api/pith-number/7WHYFJWJBUQTNSFRM4WHRHFB3S/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/7WHYFJWJBUQTNSFRM4WHRHFB3S/action/timestamp_anchor","attest_storage":"https://pith.science/pith/7WHYFJWJBUQTNSFRM4WHRHFB3S/action/storage_attestation","attest_author":"https://pith.science/pith/7WHYFJWJBUQTNSFRM4WHRHFB3S/action/author_attestation","sign_citation":"https://pith.science/pith/7WHYFJWJBUQTNSFRM4WHRHFB3S/action/citation_signature","submit_replication":"https://pith.science/pith/7WHYFJWJBUQTNSFRM4WHRHFB3S/action/replication_record"}},"created_at":"2026-07-05T11:10:05.747109+00:00","updated_at":"2026-07-05T11:10:05.747109+00:00"}