{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:ENVLPO4QPJG3KLIPIKXR5XB4PM","short_pith_number":"pith:ENVLPO4Q","schema_version":"1.0","canonical_sha256":"236ab7bb907a4db52d0f42af1edc3c7b116d2d71a58a0090f9c745a9e83337ef","source":{"kind":"arxiv","id":"2402.01742","version":1},"attestation_state":"computed","paper":{"title":"Towards Optimizing the Costs of LLM Usage","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Apoorv Saxena, Atharv Tyagi, Koyel Mukherjee, Nishanth Kotla, Shivanshu Shekhar, Tanishq Dubey","submitted_at":"2024-01-29T16:36:31Z","abstract_excerpt":"Generative AI and LLMs in particular are heavily used nowadays for various document processing tasks such as question answering and summarization. However, different LLMs come with different capabilities for different tasks as well as with different costs, tokenization, and latency. In fact, enterprises are already incurring huge costs of operating or using LLMs for their respective use cases.\n  In this work, we propose optimizing the usage costs of LLMs by estimating their output quality (without actually invoking the LLMs), and then solving an optimization routine for the LLM selection to ei"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2402.01742","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2024-01-29T16:36:31Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"4f48ec066e48596a9138442e762683b0b229627df1ae5158b636ce09f35f0778","abstract_canon_sha256":"31d329eeb13da515c1d32ca46b593fbf21cd0bdf65fd5b2f43bd18a8f6930f35"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:41:05.978788Z","signature_b64":"0CYfp7Y4bZthlAutKl3givyfHSTWtxw8ol6rDNkvXIfCL3WAWbaYJztV6/JEugqdSSQBCVd1yajyki6wiktsCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"236ab7bb907a4db52d0f42af1edc3c7b116d2d71a58a0090f9c745a9e83337ef","last_reissued_at":"2026-07-05T07:41:05.978251Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:41:05.978251Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Towards Optimizing the Costs of LLM Usage","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Apoorv Saxena, Atharv Tyagi, Koyel Mukherjee, Nishanth Kotla, Shivanshu Shekhar, Tanishq Dubey","submitted_at":"2024-01-29T16:36:31Z","abstract_excerpt":"Generative AI and LLMs in particular are heavily used nowadays for various document processing tasks such as question answering and summarization. However, different LLMs come with different capabilities for different tasks as well as with different costs, tokenization, and latency. In fact, enterprises are already incurring huge costs of operating or using LLMs for their respective use cases.\n  In this work, we propose optimizing the usage costs of LLMs by estimating their output quality (without actually invoking the LLMs), and then solving an optimization routine for the LLM selection to ei"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2402.01742","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2402.01742/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2402.01742","created_at":"2026-07-05T07:41:05.978309+00:00"},{"alias_kind":"arxiv_version","alias_value":"2402.01742v1","created_at":"2026-07-05T07:41:05.978309+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2402.01742","created_at":"2026-07-05T07:41:05.978309+00:00"},{"alias_kind":"pith_short_12","alias_value":"ENVLPO4QPJG3","created_at":"2026-07-05T07:41:05.978309+00:00"},{"alias_kind":"pith_short_16","alias_value":"ENVLPO4QPJG3KLIP","created_at":"2026-07-05T07:41:05.978309+00:00"},{"alias_kind":"pith_short_8","alias_value":"ENVLPO4Q","created_at":"2026-07-05T07:41:05.978309+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.24194","citing_title":"Dialogue to Discovery: Attribute-Aware Preference Elicitation for Conversational Product Search Assistants","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2606.28365","citing_title":"CAMI: Cost-Aware Agent-Guided Multi-Indexing for Semantic Retrieval","ref_index":44,"is_internal_anchor":false},{"citing_arxiv_id":"2504.06307","citing_title":"Optimizing Large Language Models: Metrics, Energy Efficiency, and Case Study Insights","ref_index":24,"is_internal_anchor":false},{"citing_arxiv_id":"2506.19045","citing_title":"Efficient Black-Box Fault Localization for System-Level Test Code Using Large Language Models","ref_index":35,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ENVLPO4QPJG3KLIPIKXR5XB4PM","json":"https://pith.science/pith/ENVLPO4QPJG3KLIPIKXR5XB4PM.json","graph_json":"https://pith.science/api/pith-number/ENVLPO4QPJG3KLIPIKXR5XB4PM/graph.json","events_json":"https://pith.science/api/pith-number/ENVLPO4QPJG3KLIPIKXR5XB4PM/events.json","paper":"https://pith.science/paper/ENVLPO4Q"},"agent_actions":{"view_html":"https://pith.science/pith/ENVLPO4QPJG3KLIPIKXR5XB4PM","download_json":"https://pith.science/pith/ENVLPO4QPJG3KLIPIKXR5XB4PM.json","view_paper":"https://pith.science/paper/ENVLPO4Q","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2402.01742&json=true","fetch_graph":"https://pith.science/api/pith-number/ENVLPO4QPJG3KLIPIKXR5XB4PM/graph.json","fetch_events":"https://pith.science/api/pith-number/ENVLPO4QPJG3KLIPIKXR5XB4PM/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ENVLPO4QPJG3KLIPIKXR5XB4PM/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ENVLPO4QPJG3KLIPIKXR5XB4PM/action/storage_attestation","attest_author":"https://pith.science/pith/ENVLPO4QPJG3KLIPIKXR5XB4PM/action/author_attestation","sign_citation":"https://pith.science/pith/ENVLPO4QPJG3KLIPIKXR5XB4PM/action/citation_signature","submit_replication":"https://pith.science/pith/ENVLPO4QPJG3KLIPIKXR5XB4PM/action/replication_record"}},"created_at":"2026-07-05T07:41:05.978309+00:00","updated_at":"2026-07-05T07:41:05.978309+00:00"}