{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:SQI556FQNWBTWO7EHULQLZ2DWQ","short_pith_number":"pith:SQI556FQ","schema_version":"1.0","canonical_sha256":"9411def8b06d833b3be43d1705e743b419a0b251c23690f2e174904f5d9dc37e","source":{"kind":"arxiv","id":"2506.22716","version":1},"attestation_state":"computed","paper":{"title":"BEST-Route: Adaptive LLM Routing with Test-Time Optimal Compute","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.DB"],"primary_cat":"cs.LG","authors_text":"Ankur Mallick, Chi Wang, Daniel Madrigal, Dujian Ding, Laks V.S. Lakshmanan, Menglin Xia, Mirian Del Carmen Hipolito Garcia, Qingyun Wu, Shaokun Zhang, Victor R\\\"uhle","submitted_at":"2025-06-28T01:52:50Z","abstract_excerpt":"Large language models (LLMs) are powerful tools but are often expensive to deploy at scale. LLM query routing mitigates this by dynamically assigning queries to models of varying cost and quality to obtain a desired trade-off. Prior query routing approaches generate only one response from the selected model and a single response from a small (inexpensive) model was often not good enough to beat a response from a large (expensive) model due to which they end up overusing the large model and missing out on potential cost savings. However, it is well known that for small models, generating multip"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2506.22716","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.LG","submitted_at":"2025-06-28T01:52:50Z","cross_cats_sorted":["cs.AI","cs.CL","cs.DB"],"title_canon_sha256":"84b7d33dde7da4bb192368e8993d2787ca86cfd228e2fc3f064960b8251fd8c2","abstract_canon_sha256":"3adef21e301cdde44d00d55d0bfc87277586772f781fdfd0bba63a7eebf97d8e"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:28:44.096422Z","signature_b64":"3Y0UdOW6jvJ3qtHp+Gv5cym8boGWbypMy9lIW6IgvIgwF45CibyWSPTM0Sgjz6NkdZBrHtlbhesZaBEKaoL8Bg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"9411def8b06d833b3be43d1705e743b419a0b251c23690f2e174904f5d9dc37e","last_reissued_at":"2026-07-05T11:28:44.095912Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:28:44.095912Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"BEST-Route: Adaptive LLM Routing with Test-Time Optimal Compute","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.DB"],"primary_cat":"cs.LG","authors_text":"Ankur Mallick, Chi Wang, Daniel Madrigal, Dujian Ding, Laks V.S. Lakshmanan, Menglin Xia, Mirian Del Carmen Hipolito Garcia, Qingyun Wu, Shaokun Zhang, Victor R\\\"uhle","submitted_at":"2025-06-28T01:52:50Z","abstract_excerpt":"Large language models (LLMs) are powerful tools but are often expensive to deploy at scale. LLM query routing mitigates this by dynamically assigning queries to models of varying cost and quality to obtain a desired trade-off. Prior query routing approaches generate only one response from the selected model and a single response from a small (inexpensive) model was often not good enough to beat a response from a large (expensive) model due to which they end up overusing the large model and missing out on potential cost savings. However, it is well known that for small models, generating multip"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.22716","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2506.22716/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2506.22716","created_at":"2026-07-05T11:28:44.095971+00:00"},{"alias_kind":"arxiv_version","alias_value":"2506.22716v1","created_at":"2026-07-05T11:28:44.095971+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.22716","created_at":"2026-07-05T11:28:44.095971+00:00"},{"alias_kind":"pith_short_12","alias_value":"SQI556FQNWBT","created_at":"2026-07-05T11:28:44.095971+00:00"},{"alias_kind":"pith_short_16","alias_value":"SQI556FQNWBTWO7E","created_at":"2026-07-05T11:28:44.095971+00:00"},{"alias_kind":"pith_short_8","alias_value":"SQI556FQ","created_at":"2026-07-05T11:28:44.095971+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":7,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.29424","citing_title":"EntroRouter: Learning Efficient Model Routing via Entropy Regulation","ref_index":70,"is_internal_anchor":false},{"citing_arxiv_id":"2605.22873","citing_title":"When Do LLMs Reason? A Dynamical Systems View via Entropy Phase Transitions","ref_index":31,"is_internal_anchor":false},{"citing_arxiv_id":"2605.14241","citing_title":"Latency-Quality Routing for Functionally Equivalent Tools in LLM Agents","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2605.08686","citing_title":"Iterative Critique-and-Routing Controller for Multi-Agent Systems with Heterogeneous LLMs","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07180","citing_title":"Learning Agent Routing From Early Experience","ref_index":36,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07805","citing_title":"Flexible Routing via Uncertainty Decomposition","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2604.10907","citing_title":"RouterWise: Joint Resource Allocation and Routing for Latency-Aware Multi-Model LLM Serving","ref_index":8,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/SQI556FQNWBTWO7EHULQLZ2DWQ","json":"https://pith.science/pith/SQI556FQNWBTWO7EHULQLZ2DWQ.json","graph_json":"https://pith.science/api/pith-number/SQI556FQNWBTWO7EHULQLZ2DWQ/graph.json","events_json":"https://pith.science/api/pith-number/SQI556FQNWBTWO7EHULQLZ2DWQ/events.json","paper":"https://pith.science/paper/SQI556FQ"},"agent_actions":{"view_html":"https://pith.science/pith/SQI556FQNWBTWO7EHULQLZ2DWQ","download_json":"https://pith.science/pith/SQI556FQNWBTWO7EHULQLZ2DWQ.json","view_paper":"https://pith.science/paper/SQI556FQ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2506.22716&json=true","fetch_graph":"https://pith.science/api/pith-number/SQI556FQNWBTWO7EHULQLZ2DWQ/graph.json","fetch_events":"https://pith.science/api/pith-number/SQI556FQNWBTWO7EHULQLZ2DWQ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/SQI556FQNWBTWO7EHULQLZ2DWQ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/SQI556FQNWBTWO7EHULQLZ2DWQ/action/storage_attestation","attest_author":"https://pith.science/pith/SQI556FQNWBTWO7EHULQLZ2DWQ/action/author_attestation","sign_citation":"https://pith.science/pith/SQI556FQNWBTWO7EHULQLZ2DWQ/action/citation_signature","submit_replication":"https://pith.science/pith/SQI556FQNWBTWO7EHULQLZ2DWQ/action/replication_record"}},"created_at":"2026-07-05T11:28:44.095971+00:00","updated_at":"2026-07-05T11:28:44.095971+00:00"}