{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:VG2JB46E2VALJ7UQNPBXM45RKB","short_pith_number":"pith:VG2JB46E","schema_version":"1.0","canonical_sha256":"a9b490f3c4d540b4fe906bc37673b150655b09daa1bedbce595212e485f5a5ce","source":{"kind":"arxiv","id":"2310.03094","version":3},"attestation_state":"computed","paper":{"title":"Large Language Model Cascades with Mixture of Thoughts Representations for Cost-efficient Reasoning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Jie Zhao, Liang Du, Min Zhang, Murong Yue, Ziyu Yao","submitted_at":"2023-10-04T18:21:17Z","abstract_excerpt":"Large language models (LLMs) such as GPT-4 have exhibited remarkable performance in a variety of tasks, but this strong performance often comes with the high expense of using paid API services. In this paper, we are motivated to study building an LLM cascade to save the cost of using LLMs, particularly for performing reasoning (e.g., mathematical, causal) tasks. Our cascade pipeline follows the intuition that simpler questions can be addressed by a weaker but more affordable LLM, whereas only the challenging questions necessitate the stronger and more expensive LLM. To realize this decision-ma"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2310.03094","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2023-10-04T18:21:17Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"83cb667a91a7253cdec05beeb41ed179a6473baf9ec93c5ce809df8efeaa259f","abstract_canon_sha256":"c4aad6dcdbe7e909fc68a2292b0762e99ded02d1c251309761585735402e8117"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:43:12.708145Z","signature_b64":"3Fzcn3JXg4lczqQFtfSypTWtcI2EQ8ny8vA5UFamdudpCGg1AKrv412GtMUL0imkBb55N2s2hQo04KX6KUgpAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"a9b490f3c4d540b4fe906bc37673b150655b09daa1bedbce595212e485f5a5ce","last_reissued_at":"2026-07-05T07:43:12.707613Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:43:12.707613Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Large Language Model Cascades with Mixture of Thoughts Representations for Cost-efficient Reasoning","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Jie Zhao, Liang Du, Min Zhang, Murong Yue, Ziyu Yao","submitted_at":"2023-10-04T18:21:17Z","abstract_excerpt":"Large language models (LLMs) such as GPT-4 have exhibited remarkable performance in a variety of tasks, but this strong performance often comes with the high expense of using paid API services. In this paper, we are motivated to study building an LLM cascade to save the cost of using LLMs, particularly for performing reasoning (e.g., mathematical, causal) tasks. Our cascade pipeline follows the intuition that simpler questions can be addressed by a weaker but more affordable LLM, whereas only the challenging questions necessitate the stronger and more expensive LLM. To realize this decision-ma"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2310.03094","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2310.03094/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2310.03094","created_at":"2026-07-05T07:43:12.707675+00:00"},{"alias_kind":"arxiv_version","alias_value":"2310.03094v3","created_at":"2026-07-05T07:43:12.707675+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2310.03094","created_at":"2026-07-05T07:43:12.707675+00:00"},{"alias_kind":"pith_short_12","alias_value":"VG2JB46E2VAL","created_at":"2026-07-05T07:43:12.707675+00:00"},{"alias_kind":"pith_short_16","alias_value":"VG2JB46E2VALJ7UQ","created_at":"2026-07-05T07:43:12.707675+00:00"},{"alias_kind":"pith_short_8","alias_value":"VG2JB46E","created_at":"2026-07-05T07:43:12.707675+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":13,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.22840","citing_title":"RLM-Cascade: Response-Level Speculative Decoding for Cost-Efficient LLM API Serving","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2606.20295","citing_title":"Token-Operations-Oriented Inference Optimization Techniques for Large Models","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2606.18774","citing_title":"RouteJudge: An Open Platform for Reproducible and Preference-Aware LLM Routing","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2606.06924","citing_title":"From Sampled Outcomes to Capability Distributions: Rethinking Supervision for LLM Routing","ref_index":151,"is_internal_anchor":false},{"citing_arxiv_id":"2410.15761","citing_title":"Optimal Query Allocation in Extractive QA with LLMs: A Learning-to-Defer Framework with Theoretical Guarantees","ref_index":45,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12340","citing_title":"Online Learning-to-Defer with Varying Experts","ref_index":96,"is_internal_anchor":false},{"citing_arxiv_id":"2604.23477","citing_title":"SEMA-SQL: Beyond Traditional Relational Querying with Large Language Models","ref_index":57,"is_internal_anchor":false},{"citing_arxiv_id":"2604.02367","citing_title":"Evaluating Small Language Models for Front-Door Routing: A Harmonized Benchmark and Synthetic-Traffic Experiment","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2603.27098","citing_title":"Ensemble-Based Uncertainty Estimation for Code Correctness Estimation","ref_index":48,"is_internal_anchor":false},{"citing_arxiv_id":"2605.12340","citing_title":"Online Learning-to-Defer with Varying Experts","ref_index":96,"is_internal_anchor":false},{"citing_arxiv_id":"2604.23477","citing_title":"SEMA-SQL: Beyond Traditional Relational Querying with Large Language Models","ref_index":57,"is_internal_anchor":false},{"citing_arxiv_id":"2604.18612","citing_title":"Agent-GWO: Collaborative Agents for Dynamic Prompt Optimization in Large Language Models","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2604.15728","citing_title":"Privacy-Preserving LLMs Routing","ref_index":19,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/VG2JB46E2VALJ7UQNPBXM45RKB","json":"https://pith.science/pith/VG2JB46E2VALJ7UQNPBXM45RKB.json","graph_json":"https://pith.science/api/pith-number/VG2JB46E2VALJ7UQNPBXM45RKB/graph.json","events_json":"https://pith.science/api/pith-number/VG2JB46E2VALJ7UQNPBXM45RKB/events.json","paper":"https://pith.science/paper/VG2JB46E"},"agent_actions":{"view_html":"https://pith.science/pith/VG2JB46E2VALJ7UQNPBXM45RKB","download_json":"https://pith.science/pith/VG2JB46E2VALJ7UQNPBXM45RKB.json","view_paper":"https://pith.science/paper/VG2JB46E","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2310.03094&json=true","fetch_graph":"https://pith.science/api/pith-number/VG2JB46E2VALJ7UQNPBXM45RKB/graph.json","fetch_events":"https://pith.science/api/pith-number/VG2JB46E2VALJ7UQNPBXM45RKB/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/VG2JB46E2VALJ7UQNPBXM45RKB/action/timestamp_anchor","attest_storage":"https://pith.science/pith/VG2JB46E2VALJ7UQNPBXM45RKB/action/storage_attestation","attest_author":"https://pith.science/pith/VG2JB46E2VALJ7UQNPBXM45RKB/action/author_attestation","sign_citation":"https://pith.science/pith/VG2JB46E2VALJ7UQNPBXM45RKB/action/citation_signature","submit_replication":"https://pith.science/pith/VG2JB46E2VALJ7UQNPBXM45RKB/action/replication_record"}},"created_at":"2026-07-05T07:43:12.707675+00:00","updated_at":"2026-07-05T07:43:12.707675+00:00"}