{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:DQNHWRUVSCSM4U5XEVZNTUPUJE","short_pith_number":"pith:DQNHWRUV","schema_version":"1.0","canonical_sha256":"1c1a7b469590a4ce53b72572d9d1f4490de6d3599e7d74aa39ab3d0e491686bb","source":{"kind":"arxiv","id":"2403.20306","version":1},"attestation_state":"computed","paper":{"title":"Towards Greener LLMs: Bringing Energy-Efficiency to the Forefront of LLM Inference","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.AR","cs.DC"],"primary_cat":"cs.AI","authors_text":"Chaojie Zhang, Esha Choukse, Inigo Goiri, Josep Torrellas, Jovan Stojkovic","submitted_at":"2024-03-29T17:22:48Z","abstract_excerpt":"With the ubiquitous use of modern large language models (LLMs) across industries, the inference serving for these models is ever expanding. Given the high compute and memory requirements of modern LLMs, more and more top-of-the-line GPUs are being deployed to serve these models. Energy availability has come to the forefront as the biggest challenge for data center expansion to serve these models. In this paper, we present the trade-offs brought up by making energy efficiency the primary goal of LLM serving under performance SLOs. We show that depending on the inputs, the model, and the service"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2403.20306","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.AI","submitted_at":"2024-03-29T17:22:48Z","cross_cats_sorted":["cs.AR","cs.DC"],"title_canon_sha256":"4e56a7c212eceb658fb16eff91a241fb6f30b88963d0e498e513cd09f6f05088","abstract_canon_sha256":"a7e28ade39d9a9cef6270a3178283989e3c3946833b525590b9be6c7e08a1eb7"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:02:17.387620Z","signature_b64":"pzJptXDPkaHg+gquzy9hR1RMi+hsTW4s9aQMIh1wTkyQn1HX6hZJ3p5NbK73uAJVKbjtBQLGKeioC5C39bQqCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"1c1a7b469590a4ce53b72572d9d1f4490de6d3599e7d74aa39ab3d0e491686bb","last_reissued_at":"2026-07-05T08:02:17.387155Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:02:17.387155Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Towards Greener LLMs: Bringing Energy-Efficiency to the Forefront of LLM Inference","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.AR","cs.DC"],"primary_cat":"cs.AI","authors_text":"Chaojie Zhang, Esha Choukse, Inigo Goiri, Josep Torrellas, Jovan Stojkovic","submitted_at":"2024-03-29T17:22:48Z","abstract_excerpt":"With the ubiquitous use of modern large language models (LLMs) across industries, the inference serving for these models is ever expanding. Given the high compute and memory requirements of modern LLMs, more and more top-of-the-line GPUs are being deployed to serve these models. Energy availability has come to the forefront as the biggest challenge for data center expansion to serve these models. In this paper, we present the trade-offs brought up by making energy efficiency the primary goal of LLM serving under performance SLOs. We show that depending on the inputs, the model, and the service"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2403.20306","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2403.20306/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2403.20306","created_at":"2026-07-05T08:02:17.387208+00:00"},{"alias_kind":"arxiv_version","alias_value":"2403.20306v1","created_at":"2026-07-05T08:02:17.387208+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2403.20306","created_at":"2026-07-05T08:02:17.387208+00:00"},{"alias_kind":"pith_short_12","alias_value":"DQNHWRUVSCSM","created_at":"2026-07-05T08:02:17.387208+00:00"},{"alias_kind":"pith_short_16","alias_value":"DQNHWRUVSCSM4U5X","created_at":"2026-07-05T08:02:17.387208+00:00"},{"alias_kind":"pith_short_8","alias_value":"DQNHWRUV","created_at":"2026-07-05T08:02:17.387208+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":8,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.23001","citing_title":"EnerInfer: Energy-Aware On-Device LLM Inference","ref_index":55,"is_internal_anchor":false},{"citing_arxiv_id":"2605.29641","citing_title":"Experimentation for Different Scheduling Policies on Queues: Mixed Differences-in-Q Estimators Based on Little's Law","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2503.10666","citing_title":"Green Prompting: Characterizing Prompt-driven Energy Costs of LLM Inference","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2602.05695","citing_title":"SweetSpot: An Analytical Model for Predicting Energy Efficiency of LLM Inference","ref_index":20,"is_internal_anchor":false},{"citing_arxiv_id":"2602.17697","citing_title":"Pimp My LLM: Leveraging Variability Modeling to Tune Inference Hyperparameters","ref_index":59,"is_internal_anchor":false},{"citing_arxiv_id":"2602.18755","citing_title":"DualScale: Energy-Efficient Disaggregated LLM Serving via Phase-Aware Placement and DVFS","ref_index":42,"is_internal_anchor":false},{"citing_arxiv_id":"2604.10769","citing_title":"Workload composition smooths aggregate power demand while sustaining short-horizon ramps in AI data centers","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2604.16682","citing_title":"KAIROS: Stateful, Context-Aware Power-Efficient Agentic Inference Serving","ref_index":60,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/DQNHWRUVSCSM4U5XEVZNTUPUJE","json":"https://pith.science/pith/DQNHWRUVSCSM4U5XEVZNTUPUJE.json","graph_json":"https://pith.science/api/pith-number/DQNHWRUVSCSM4U5XEVZNTUPUJE/graph.json","events_json":"https://pith.science/api/pith-number/DQNHWRUVSCSM4U5XEVZNTUPUJE/events.json","paper":"https://pith.science/paper/DQNHWRUV"},"agent_actions":{"view_html":"https://pith.science/pith/DQNHWRUVSCSM4U5XEVZNTUPUJE","download_json":"https://pith.science/pith/DQNHWRUVSCSM4U5XEVZNTUPUJE.json","view_paper":"https://pith.science/paper/DQNHWRUV","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2403.20306&json=true","fetch_graph":"https://pith.science/api/pith-number/DQNHWRUVSCSM4U5XEVZNTUPUJE/graph.json","fetch_events":"https://pith.science/api/pith-number/DQNHWRUVSCSM4U5XEVZNTUPUJE/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/DQNHWRUVSCSM4U5XEVZNTUPUJE/action/timestamp_anchor","attest_storage":"https://pith.science/pith/DQNHWRUVSCSM4U5XEVZNTUPUJE/action/storage_attestation","attest_author":"https://pith.science/pith/DQNHWRUVSCSM4U5XEVZNTUPUJE/action/author_attestation","sign_citation":"https://pith.science/pith/DQNHWRUVSCSM4U5XEVZNTUPUJE/action/citation_signature","submit_replication":"https://pith.science/pith/DQNHWRUVSCSM4U5XEVZNTUPUJE/action/replication_record"}},"created_at":"2026-07-05T08:02:17.387208+00:00","updated_at":"2026-07-05T08:02:17.387208+00:00"}