{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:CW2DRDYU4SKKOCHRX4XBLSR2ZL","short_pith_number":"pith:CW2DRDYU","schema_version":"1.0","canonical_sha256":"15b4388f14e494a708f1bf2e15ca3acadd40c4e5e685e6f3e582c3a4cc382c5f","source":{"kind":"arxiv","id":"2305.01879","version":4},"attestation_state":"computed","paper":{"title":"SCOTT: Self-Consistent Chain-of-Thought Distillation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Bing Yin, Peifeng Wang, Xiang Ren, Yifan Gao, Zheng Li, Zhengyang Wang","submitted_at":"2023-05-03T03:47:00Z","abstract_excerpt":"Large language models (LMs) beyond a certain scale, demonstrate the emergent capability of generating free-text rationales for their predictions via chain-of-thought (CoT) prompting. While CoT can yield dramatically improved performance, such gains are only observed for sufficiently large LMs. Even more concerning, there is little guarantee that the generated rationales are consistent with LM's predictions or faithfully justify the decisions. In this work, we propose a faithful knowledge distillation method to learn a small, self-consistent CoT model from a teacher model that is orders of magn"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2305.01879","kind":"arxiv","version":4},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2023-05-03T03:47:00Z","cross_cats_sorted":[],"title_canon_sha256":"108a06ea397a67835dc53e10d0112ad61794a5596ea776f3ea1829b5b21a2415","abstract_canon_sha256":"3de214db7396ca582d29a035dd570342f86d03765d042582d64ddbd63e5cccb1"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:46:22.321423Z","signature_b64":"/JZwkWgPU5k8Fo4WF4376WqF7WzLIh6cKJQW+eqjB5/m7Izk1o+o7yDnOQYRXLlHuWiXHY7g0CuXBQVaj0U3CA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"15b4388f14e494a708f1bf2e15ca3acadd40c4e5e685e6f3e582c3a4cc382c5f","last_reissued_at":"2026-07-05T06:46:22.320885Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:46:22.320885Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"SCOTT: Self-Consistent Chain-of-Thought Distillation","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Bing Yin, Peifeng Wang, Xiang Ren, Yifan Gao, Zheng Li, Zhengyang Wang","submitted_at":"2023-05-03T03:47:00Z","abstract_excerpt":"Large language models (LMs) beyond a certain scale, demonstrate the emergent capability of generating free-text rationales for their predictions via chain-of-thought (CoT) prompting. While CoT can yield dramatically improved performance, such gains are only observed for sufficiently large LMs. Even more concerning, there is little guarantee that the generated rationales are consistent with LM's predictions or faithfully justify the decisions. In this work, we propose a faithful knowledge distillation method to learn a small, self-consistent CoT model from a teacher model that is orders of magn"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2305.01879","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2305.01879/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2305.01879","created_at":"2026-07-05T06:46:22.320961+00:00"},{"alias_kind":"arxiv_version","alias_value":"2305.01879v4","created_at":"2026-07-05T06:46:22.320961+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2305.01879","created_at":"2026-07-05T06:46:22.320961+00:00"},{"alias_kind":"pith_short_12","alias_value":"CW2DRDYU4SKK","created_at":"2026-07-05T06:46:22.320961+00:00"},{"alias_kind":"pith_short_16","alias_value":"CW2DRDYU4SKKOCHR","created_at":"2026-07-05T06:46:22.320961+00:00"},{"alias_kind":"pith_short_8","alias_value":"CW2DRDYU","created_at":"2026-07-05T06:46:22.320961+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.06840","citing_title":"Characterize Then Distill: Mechanistic Reasoning in Large Output Spaces","ref_index":158,"is_internal_anchor":false},{"citing_arxiv_id":"2601.13992","citing_title":"\"The Whole Is Greater Than the Sum of Its Parts\": A Compatibility-Aware Multi-Teacher CoT Distillation Framework","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2404.14294","citing_title":"A Survey on Efficient Inference for Large Language Models","ref_index":127,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08842","citing_title":"XPERT: Expert Knowledge Transfer for Effective Training of Language Models","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07783","citing_title":"Chain-based Distillation for Effective Initialization of Variable-Sized Small Language Models","ref_index":99,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/CW2DRDYU4SKKOCHRX4XBLSR2ZL","json":"https://pith.science/pith/CW2DRDYU4SKKOCHRX4XBLSR2ZL.json","graph_json":"https://pith.science/api/pith-number/CW2DRDYU4SKKOCHRX4XBLSR2ZL/graph.json","events_json":"https://pith.science/api/pith-number/CW2DRDYU4SKKOCHRX4XBLSR2ZL/events.json","paper":"https://pith.science/paper/CW2DRDYU"},"agent_actions":{"view_html":"https://pith.science/pith/CW2DRDYU4SKKOCHRX4XBLSR2ZL","download_json":"https://pith.science/pith/CW2DRDYU4SKKOCHRX4XBLSR2ZL.json","view_paper":"https://pith.science/paper/CW2DRDYU","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2305.01879&json=true","fetch_graph":"https://pith.science/api/pith-number/CW2DRDYU4SKKOCHRX4XBLSR2ZL/graph.json","fetch_events":"https://pith.science/api/pith-number/CW2DRDYU4SKKOCHRX4XBLSR2ZL/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/CW2DRDYU4SKKOCHRX4XBLSR2ZL/action/timestamp_anchor","attest_storage":"https://pith.science/pith/CW2DRDYU4SKKOCHRX4XBLSR2ZL/action/storage_attestation","attest_author":"https://pith.science/pith/CW2DRDYU4SKKOCHRX4XBLSR2ZL/action/author_attestation","sign_citation":"https://pith.science/pith/CW2DRDYU4SKKOCHRX4XBLSR2ZL/action/citation_signature","submit_replication":"https://pith.science/pith/CW2DRDYU4SKKOCHRX4XBLSR2ZL/action/replication_record"}},"created_at":"2026-07-05T06:46:22.320961+00:00","updated_at":"2026-07-05T06:46:22.320961+00:00"}