{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:ZDJW2D3GA6GCHKVGIM5KS4ECK4","short_pith_number":"pith:ZDJW2D3G","schema_version":"1.0","canonical_sha256":"c8d36d0f66078c23aaa6433aa97082573203070d7256ac5bf681ec605ba710e7","source":{"kind":"arxiv","id":"2506.06579","version":1},"attestation_state":"computed","paper":{"title":"Towards Efficient Multi-LLM Inference: Characterization and Analysis of LLM Routing and Hierarchical Techniques","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.DC"],"primary_cat":"cs.LG","authors_text":"Adarsh Prasad Behera, James Gross, Jaya Prakash Champati, Roberto Morabito, Sasu Tarkoma","submitted_at":"2025-06-06T23:13:08Z","abstract_excerpt":"Recent progress in Language Models (LMs) has dramatically advanced the field of natural language processing (NLP), excelling at tasks like text generation, summarization, and question answering. However, their inference remains computationally expensive and energy intensive, especially in settings with limited hardware, power, or bandwidth. This makes it difficult to deploy LMs in mobile, edge, or cost sensitive environments. To address these challenges, recent approaches have introduced multi LLM intelligent model selection strategies that dynamically allocate computational resources based on"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2506.06579","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.LG","submitted_at":"2025-06-06T23:13:08Z","cross_cats_sorted":["cs.AI","cs.CL","cs.DC"],"title_canon_sha256":"b99cba5434b5a9ac3d92ee4e7b994408e8f48222e67a95ac60c26ea7d37353b6","abstract_canon_sha256":"2909853255a8afc4b7ae989304ad739fca55d126282af1f67df51ef01dc86fb6"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:17:52.340888Z","signature_b64":"UMM4i6hO7jQrRAdVYayi6xW10l3sxaqZOk+HzYm/Ou0VLfKIJFOU3BZ8tSExXhDfO3y4F9AMGH9eXR6XFXxPDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"c8d36d0f66078c23aaa6433aa97082573203070d7256ac5bf681ec605ba710e7","last_reissued_at":"2026-07-05T11:17:52.340462Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:17:52.340462Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Towards Efficient Multi-LLM Inference: Characterization and Analysis of LLM Routing and Hierarchical Techniques","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.DC"],"primary_cat":"cs.LG","authors_text":"Adarsh Prasad Behera, James Gross, Jaya Prakash Champati, Roberto Morabito, Sasu Tarkoma","submitted_at":"2025-06-06T23:13:08Z","abstract_excerpt":"Recent progress in Language Models (LMs) has dramatically advanced the field of natural language processing (NLP), excelling at tasks like text generation, summarization, and question answering. However, their inference remains computationally expensive and energy intensive, especially in settings with limited hardware, power, or bandwidth. This makes it difficult to deploy LMs in mobile, edge, or cost sensitive environments. To address these challenges, recent approaches have introduced multi LLM intelligent model selection strategies that dynamically allocate computational resources based on"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2506.06579","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2506.06579/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2506.06579","created_at":"2026-07-05T11:17:52.340519+00:00"},{"alias_kind":"arxiv_version","alias_value":"2506.06579v1","created_at":"2026-07-05T11:17:52.340519+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2506.06579","created_at":"2026-07-05T11:17:52.340519+00:00"},{"alias_kind":"pith_short_12","alias_value":"ZDJW2D3GA6GC","created_at":"2026-07-05T11:17:52.340519+00:00"},{"alias_kind":"pith_short_16","alias_value":"ZDJW2D3GA6GCHKVG","created_at":"2026-07-05T11:17:52.340519+00:00"},{"alias_kind":"pith_short_8","alias_value":"ZDJW2D3G","created_at":"2026-07-05T11:17:52.340519+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.06924","citing_title":"From Sampled Outcomes to Capability Distributions: Rethinking Supervision for LLM Routing","ref_index":147,"is_internal_anchor":false},{"citing_arxiv_id":"2606.06840","citing_title":"Characterize Then Distill: Mechanistic Reasoning in Large Output Spaces","ref_index":141,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11317","citing_title":"SOMA: Efficient Multi-turn LLM Serving via Small Language Model","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01111","citing_title":"When Less is Enough: Efficient Inference via Collaborative Reasoning","ref_index":2,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ZDJW2D3GA6GCHKVGIM5KS4ECK4","json":"https://pith.science/pith/ZDJW2D3GA6GCHKVGIM5KS4ECK4.json","graph_json":"https://pith.science/api/pith-number/ZDJW2D3GA6GCHKVGIM5KS4ECK4/graph.json","events_json":"https://pith.science/api/pith-number/ZDJW2D3GA6GCHKVGIM5KS4ECK4/events.json","paper":"https://pith.science/paper/ZDJW2D3G"},"agent_actions":{"view_html":"https://pith.science/pith/ZDJW2D3GA6GCHKVGIM5KS4ECK4","download_json":"https://pith.science/pith/ZDJW2D3GA6GCHKVGIM5KS4ECK4.json","view_paper":"https://pith.science/paper/ZDJW2D3G","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2506.06579&json=true","fetch_graph":"https://pith.science/api/pith-number/ZDJW2D3GA6GCHKVGIM5KS4ECK4/graph.json","fetch_events":"https://pith.science/api/pith-number/ZDJW2D3GA6GCHKVGIM5KS4ECK4/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ZDJW2D3GA6GCHKVGIM5KS4ECK4/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ZDJW2D3GA6GCHKVGIM5KS4ECK4/action/storage_attestation","attest_author":"https://pith.science/pith/ZDJW2D3GA6GCHKVGIM5KS4ECK4/action/author_attestation","sign_citation":"https://pith.science/pith/ZDJW2D3GA6GCHKVGIM5KS4ECK4/action/citation_signature","submit_replication":"https://pith.science/pith/ZDJW2D3GA6GCHKVGIM5KS4ECK4/action/replication_record"}},"created_at":"2026-07-05T11:17:52.340519+00:00","updated_at":"2026-07-05T11:17:52.340519+00:00"}