{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:UZQXN5UT5ANLMQ5YD3XZFGYVRF","short_pith_number":"pith:UZQXN5UT","schema_version":"1.0","canonical_sha256":"a66176f693e81ab643b81eef929b15895c1466f00b9a0b685cfef94d2be52a67","source":{"kind":"arxiv","id":"2210.06726","version":1},"attestation_state":"computed","paper":{"title":"Explanations from Large Language Models Make Small Reasoners Better","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Baolin Peng, Hong Wang, Jianshu Chen, Jing Qian, Shiyang Li, Wenhu Chen, Xifeng Yan, Xinlu Zhang, Yelong Shen, Yi Mao, Zekun Li, Zhiyu Chen","submitted_at":"2022-10-13T04:50:02Z","abstract_excerpt":"Integrating free-text explanations to in-context learning of large language models (LLM) is shown to elicit strong reasoning capabilities along with reasonable explanations. In this paper, we consider the problem of leveraging the explanations generated by LLM to improve the training of small reasoners, which are more favorable in real-production deployment due to their low cost. We systematically explore three explanation generation approaches from LLM and utilize a multi-task learning framework to facilitate small models to acquire strong reasoning power together with explanation generation "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2210.06726","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2022-10-13T04:50:02Z","cross_cats_sorted":[],"title_canon_sha256":"aaf9238538f307edac02b509e8c57e17a77a184f50d1364ac06e5392a5a3a001","abstract_canon_sha256":"0006c063d5354527315b404cbad8f90160fbc02b2eb1ec8fb7e1b8c22c16d62a"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:06:19.759462Z","signature_b64":"NqJlkN+iC7p5zJNLRRjj/gc4dpWRVWsR65n1bhAMkHamPZ4yPqt8xlOuu2qtP5oIxPVA3+xLWuzZ2gvc2cxQBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"a66176f693e81ab643b81eef929b15895c1466f00b9a0b685cfef94d2be52a67","last_reissued_at":"2026-07-05T05:06:19.758966Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:06:19.758966Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Explanations from Large Language Models Make Small Reasoners Better","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Baolin Peng, Hong Wang, Jianshu Chen, Jing Qian, Shiyang Li, Wenhu Chen, Xifeng Yan, Xinlu Zhang, Yelong Shen, Yi Mao, Zekun Li, Zhiyu Chen","submitted_at":"2022-10-13T04:50:02Z","abstract_excerpt":"Integrating free-text explanations to in-context learning of large language models (LLM) is shown to elicit strong reasoning capabilities along with reasonable explanations. In this paper, we consider the problem of leveraging the explanations generated by LLM to improve the training of small reasoners, which are more favorable in real-production deployment due to their low cost. We systematically explore three explanation generation approaches from LLM and utilize a multi-task learning framework to facilitate small models to acquire strong reasoning power together with explanation generation "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2210.06726","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2210.06726/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2210.06726","created_at":"2026-07-05T05:06:19.759025+00:00"},{"alias_kind":"arxiv_version","alias_value":"2210.06726v1","created_at":"2026-07-05T05:06:19.759025+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2210.06726","created_at":"2026-07-05T05:06:19.759025+00:00"},{"alias_kind":"pith_short_12","alias_value":"UZQXN5UT5ANL","created_at":"2026-07-05T05:06:19.759025+00:00"},{"alias_kind":"pith_short_16","alias_value":"UZQXN5UT5ANLMQ5Y","created_at":"2026-07-05T05:06:19.759025+00:00"},{"alias_kind":"pith_short_8","alias_value":"UZQXN5UT","created_at":"2026-07-05T05:06:19.759025+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":10,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.05339","citing_title":"TREK: Distill to Explore, Reinforce to Refine","ref_index":21,"is_internal_anchor":true},{"citing_arxiv_id":"2606.12360","citing_title":"Anatomy of Post-Training: Using Interpretability to Characterize Data and Shape the Learning Signal","ref_index":57,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12483","citing_title":"Beyond GRPO and On-Policy Distillation: An Empirical Sparse-to-Dense Reward Principle for Language-Model Post-Training","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12483","citing_title":"Beyond GRPO and On-Policy Distillation: An Empirical Sparse-to-Dense Reward Principle for Language-Model Post-Training","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12483","citing_title":"Beyond GRPO and On-Policy Distillation: An Empirical Sparse-to-Dense Reward Principle for Language-Model Post-Training","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2309.12284","citing_title":"MetaMath: Bootstrap Your Own Mathematical Questions for Large Language Models","ref_index":37,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12483","citing_title":"Beyond GRPO and On-Policy Distillation: An Empirical Sparse-to-Dense Reward Principle for Language-Model Post-Training","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2605.04078","citing_title":"Validity-Calibrated Reasoning Distillation","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2307.13702","citing_title":"Measuring Faithfulness in Chain-of-Thought Reasoning","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2605.04078","citing_title":"Validity-Calibrated Reasoning Distillation","ref_index":36,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/UZQXN5UT5ANLMQ5YD3XZFGYVRF","json":"https://pith.science/pith/UZQXN5UT5ANLMQ5YD3XZFGYVRF.json","graph_json":"https://pith.science/api/pith-number/UZQXN5UT5ANLMQ5YD3XZFGYVRF/graph.json","events_json":"https://pith.science/api/pith-number/UZQXN5UT5ANLMQ5YD3XZFGYVRF/events.json","paper":"https://pith.science/paper/UZQXN5UT"},"agent_actions":{"view_html":"https://pith.science/pith/UZQXN5UT5ANLMQ5YD3XZFGYVRF","download_json":"https://pith.science/pith/UZQXN5UT5ANLMQ5YD3XZFGYVRF.json","view_paper":"https://pith.science/paper/UZQXN5UT","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2210.06726&json=true","fetch_graph":"https://pith.science/api/pith-number/UZQXN5UT5ANLMQ5YD3XZFGYVRF/graph.json","fetch_events":"https://pith.science/api/pith-number/UZQXN5UT5ANLMQ5YD3XZFGYVRF/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/UZQXN5UT5ANLMQ5YD3XZFGYVRF/action/timestamp_anchor","attest_storage":"https://pith.science/pith/UZQXN5UT5ANLMQ5YD3XZFGYVRF/action/storage_attestation","attest_author":"https://pith.science/pith/UZQXN5UT5ANLMQ5YD3XZFGYVRF/action/author_attestation","sign_citation":"https://pith.science/pith/UZQXN5UT5ANLMQ5YD3XZFGYVRF/action/citation_signature","submit_replication":"https://pith.science/pith/UZQXN5UT5ANLMQ5YD3XZFGYVRF/action/replication_record"}},"created_at":"2026-07-05T05:06:19.759025+00:00","updated_at":"2026-07-05T05:06:19.759025+00:00"}