{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:GOBKAV42AXOE3Z6PKIQHZGUFB6","short_pith_number":"pith:GOBKAV42","schema_version":"1.0","canonical_sha256":"3382a0579a05dc4de7cf52207c9a850fb8f471b3e80a106b8820a5206f247c78","source":{"kind":"arxiv","id":"2309.15789","version":1},"attestation_state":"computed","paper":{"title":"Large Language Model Routing with Benchmark Datasets","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Anthony Ou, Justin Solomon, Kate Soule, Mikhail Yurochkin, M\\'irian Silva, Neil Thompson, Tal Shnitzer, Yuekai Sun","submitted_at":"2023-09-27T17:08:40Z","abstract_excerpt":"There is a rapidly growing number of open-source Large Language Models (LLMs) and benchmark datasets to compare them. While some models dominate these benchmarks, no single model typically achieves the best accuracy in all tasks and use cases. In this work, we address the challenge of selecting the best LLM out of a collection of models for new tasks. We propose a new formulation for the problem, in which benchmark datasets are repurposed to learn a \"router\" model for this LLM selection, and we show that this problem can be reduced to a collection of binary classification tasks. We demonstrate"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2309.15789","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2023-09-27T17:08:40Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"4e82ddfd7264a2f80dadc78c32e3985009930723d1f3bd21a9659edd84557262","abstract_canon_sha256":"60b758ff3f08af7a86d94f1997462f6a4316ee82214e0cbd4d1e26b4d5058df6"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:54:57.560985Z","signature_b64":"88S9oKgfJhyl8N+2kxrKtyzsJS6fXTqcF1OTJpCEj2I+xWfhMLDj98A0OUHU20rH5QKj0shiaD4JPwOCseg2CA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"3382a0579a05dc4de7cf52207c9a850fb8f471b3e80a106b8820a5206f247c78","last_reissued_at":"2026-07-05T06:54:57.560500Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:54:57.560500Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Large Language Model Routing with Benchmark Datasets","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Anthony Ou, Justin Solomon, Kate Soule, Mikhail Yurochkin, M\\'irian Silva, Neil Thompson, Tal Shnitzer, Yuekai Sun","submitted_at":"2023-09-27T17:08:40Z","abstract_excerpt":"There is a rapidly growing number of open-source Large Language Models (LLMs) and benchmark datasets to compare them. While some models dominate these benchmarks, no single model typically achieves the best accuracy in all tasks and use cases. In this work, we address the challenge of selecting the best LLM out of a collection of models for new tasks. We propose a new formulation for the problem, in which benchmark datasets are repurposed to learn a \"router\" model for this LLM selection, and we show that this problem can be reduced to a collection of binary classification tasks. We demonstrate"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2309.15789","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2309.15789/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2309.15789","created_at":"2026-07-05T06:54:57.560560+00:00"},{"alias_kind":"arxiv_version","alias_value":"2309.15789v1","created_at":"2026-07-05T06:54:57.560560+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2309.15789","created_at":"2026-07-05T06:54:57.560560+00:00"},{"alias_kind":"pith_short_12","alias_value":"GOBKAV42AXOE","created_at":"2026-07-05T06:54:57.560560+00:00"},{"alias_kind":"pith_short_16","alias_value":"GOBKAV42AXOE3Z6P","created_at":"2026-07-05T06:54:57.560560+00:00"},{"alias_kind":"pith_short_8","alias_value":"GOBKAV42","created_at":"2026-07-05T06:54:57.560560+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":19,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.05761","citing_title":"Synthetic Consumer Insight Generation with Large Language Models","ref_index":4,"is_internal_anchor":true},{"citing_arxiv_id":"2606.21929","citing_title":"Selective Ensemble Based on Preference-Directed Multi-Objective Bandits","ref_index":56,"is_internal_anchor":false},{"citing_arxiv_id":"2606.20295","citing_title":"Token-Operations-Oriented Inference Optimization Techniques for Large Models","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2604.26940","citing_title":"Select to Think: Unlocking SLM Potential with Local Sufficiency","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2605.25558","citing_title":"Beyond Query Memorization: Large Language Model Routing with Query Decomposition and Historical Matching","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2605.30736","citing_title":"OrcaRouter: A Production-Oriented LLM Router with Hybrid Offline-Online Learning","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2505.12601","citing_title":"Rethinking Predictive Modeling for LLM Routing: When Simple kNN Beats Complex Learned Routers","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2507.14200","citing_title":"A Scalable Multi-LLM Collaboration System with Retrieval-based Selection and Exploration-Exploitation-Driven Enhancement","ref_index":48,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17288","citing_title":"When Efficiency Backfires: Cascading LLMs Trigger Cascade Failure under Adversarial Attack","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2509.24814","citing_title":"A Greedy PDE Router for Blending Neural Operators and Classical Methods","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2512.23213","citing_title":"Scoring, Reasoning, and Selecting the Best! Ensembling Large Language Models via a Peer-Review Process","ref_index":41,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11317","citing_title":"SOMA: Efficient Multi-turn LLM Serving via Small Language Model","ref_index":43,"is_internal_anchor":false},{"citing_arxiv_id":"2604.26940","citing_title":"Select to Think: Unlocking SLM Potential with Local Sufficiency","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07112","citing_title":"Switchcraft: AI Model Router for Agentic Tool Calling","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07180","citing_title":"Learning Agent Routing From Early Experience","ref_index":35,"is_internal_anchor":false},{"citing_arxiv_id":"2604.06753","citing_title":"Select-then-Solve: Paradigm Routing as Inference-Time Optimization for LLM Agents","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2604.22819","citing_title":"A pragmatic approach to regulating AI agents","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2604.15728","citing_title":"Privacy-Preserving LLMs Routing","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2604.17112","citing_title":"Complementing Self-Consistency with Cross-Model Disagreement for Uncertainty Quantification","ref_index":38,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/GOBKAV42AXOE3Z6PKIQHZGUFB6","json":"https://pith.science/pith/GOBKAV42AXOE3Z6PKIQHZGUFB6.json","graph_json":"https://pith.science/api/pith-number/GOBKAV42AXOE3Z6PKIQHZGUFB6/graph.json","events_json":"https://pith.science/api/pith-number/GOBKAV42AXOE3Z6PKIQHZGUFB6/events.json","paper":"https://pith.science/paper/GOBKAV42"},"agent_actions":{"view_html":"https://pith.science/pith/GOBKAV42AXOE3Z6PKIQHZGUFB6","download_json":"https://pith.science/pith/GOBKAV42AXOE3Z6PKIQHZGUFB6.json","view_paper":"https://pith.science/paper/GOBKAV42","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2309.15789&json=true","fetch_graph":"https://pith.science/api/pith-number/GOBKAV42AXOE3Z6PKIQHZGUFB6/graph.json","fetch_events":"https://pith.science/api/pith-number/GOBKAV42AXOE3Z6PKIQHZGUFB6/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/GOBKAV42AXOE3Z6PKIQHZGUFB6/action/timestamp_anchor","attest_storage":"https://pith.science/pith/GOBKAV42AXOE3Z6PKIQHZGUFB6/action/storage_attestation","attest_author":"https://pith.science/pith/GOBKAV42AXOE3Z6PKIQHZGUFB6/action/author_attestation","sign_citation":"https://pith.science/pith/GOBKAV42AXOE3Z6PKIQHZGUFB6/action/citation_signature","submit_replication":"https://pith.science/pith/GOBKAV42AXOE3Z6PKIQHZGUFB6/action/replication_record"}},"created_at":"2026-07-05T06:54:57.560560+00:00","updated_at":"2026-07-05T06:54:57.560560+00:00"}