{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:ASXSVAGJSWJAU6ZFMKTMR5AIJA","short_pith_number":"pith:ASXSVAGJ","schema_version":"1.0","canonical_sha256":"04af2a80c995920a7b2562a6c8f408480e7d6702885ef51805a1301ab08f098e","source":{"kind":"arxiv","id":"2502.05043","version":2},"attestation_state":"computed","paper":{"title":"EcoServe: Designing Carbon-Aware AI Inference Systems","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.DC","authors_text":"Esha Choukse, G. Edward Suh, Rodrigo Fonseca, Udit Gupta, Yueying Li, Zhanqiu Hu","submitted_at":"2025-02-07T16:09:17Z","abstract_excerpt":"The rapid increase in LLM ubiquity and scale levies unprecedented demands on computing infrastructure. These demands not only incur large compute and memory resources but also significant energy, yielding large operational and embodied carbon emissions. In this work, we present three main observations based on modeling and traces from the production deployment of two Generative AI services in a major cloud service provider. First, while GPUs dominate operational carbon, host processing systems (e.g., CPUs, memory, storage) dominate embodied carbon. Second, offline, batch inference accounts for"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2502.05043","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.DC","submitted_at":"2025-02-07T16:09:17Z","cross_cats_sorted":[],"title_canon_sha256":"59e61b319e6c648a8c78a4774067363cf9a159ee39227edddf5e6ea8a5b41a45","abstract_canon_sha256":"47dab40c7ac0e207f357de65fcd48781911359378214eb065ee9384bfcce8124"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:31:53.893927Z","signature_b64":"1ypEVQjfRwZdIRWNWqHDfoMcvh6k3UdCoPJ2a04d7gF4U/AWJ7ngIjTawjmPCPKCYSrHDRbvPrfcHVnOl/4JDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"04af2a80c995920a7b2562a6c8f408480e7d6702885ef51805a1301ab08f098e","last_reissued_at":"2026-07-05T10:31:53.893444Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:31:53.893444Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"EcoServe: Designing Carbon-Aware AI Inference Systems","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.DC","authors_text":"Esha Choukse, G. Edward Suh, Rodrigo Fonseca, Udit Gupta, Yueying Li, Zhanqiu Hu","submitted_at":"2025-02-07T16:09:17Z","abstract_excerpt":"The rapid increase in LLM ubiquity and scale levies unprecedented demands on computing infrastructure. These demands not only incur large compute and memory resources but also significant energy, yielding large operational and embodied carbon emissions. In this work, we present three main observations based on modeling and traces from the production deployment of two Generative AI services in a major cloud service provider. First, while GPUs dominate operational carbon, host processing systems (e.g., CPUs, memory, storage) dominate embodied carbon. Second, offline, batch inference accounts for"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2502.05043","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2502.05043/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2502.05043","created_at":"2026-07-05T10:31:53.893495+00:00"},{"alias_kind":"arxiv_version","alias_value":"2502.05043v2","created_at":"2026-07-05T10:31:53.893495+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2502.05043","created_at":"2026-07-05T10:31:53.893495+00:00"},{"alias_kind":"pith_short_12","alias_value":"ASXSVAGJSWJA","created_at":"2026-07-05T10:31:53.893495+00:00"},{"alias_kind":"pith_short_16","alias_value":"ASXSVAGJSWJAU6ZF","created_at":"2026-07-05T10:31:53.893495+00:00"},{"alias_kind":"pith_short_8","alias_value":"ASXSVAGJ","created_at":"2026-07-05T10:31:53.893495+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":9,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.18851","citing_title":"From Tokens to Energy Flexibility: Quantization-Enabled Demand Response for Data Centers with LLM Inference Workloads","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2606.07632","citing_title":"Evaluation of ML Resource Utilization Requires Model Life Cycle Assessment","ref_index":67,"is_internal_anchor":false},{"citing_arxiv_id":"2605.27309","citing_title":"Greening AI Inference with Accuracy and Latency-aware User Incentives","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2505.23970","citing_title":"Cache Your Prompt When It's Green: Carbon-Aware Caching for Large Language Model Serving","ref_index":40,"is_internal_anchor":false},{"citing_arxiv_id":"2601.00823","citing_title":"Energy-Aware Routing to Large Reasoning Models","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2603.15042","citing_title":"Determinism-Preserving GPU Spatial Sharing with Vitamin-E","ref_index":41,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11603","citing_title":"GAR: Carbon-Aware Routing for LLM Inference via Constrained Optimization","ref_index":10,"is_internal_anchor":false},{"citing_arxiv_id":"2604.27855","citing_title":"AI Inference as Relocatable Electricity Demand: A Latency-Constrained Energy-Geography Framework","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2604.16682","citing_title":"KAIROS: Stateful, Context-Aware Power-Efficient Agentic Inference Serving","ref_index":35,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ASXSVAGJSWJAU6ZFMKTMR5AIJA","json":"https://pith.science/pith/ASXSVAGJSWJAU6ZFMKTMR5AIJA.json","graph_json":"https://pith.science/api/pith-number/ASXSVAGJSWJAU6ZFMKTMR5AIJA/graph.json","events_json":"https://pith.science/api/pith-number/ASXSVAGJSWJAU6ZFMKTMR5AIJA/events.json","paper":"https://pith.science/paper/ASXSVAGJ"},"agent_actions":{"view_html":"https://pith.science/pith/ASXSVAGJSWJAU6ZFMKTMR5AIJA","download_json":"https://pith.science/pith/ASXSVAGJSWJAU6ZFMKTMR5AIJA.json","view_paper":"https://pith.science/paper/ASXSVAGJ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2502.05043&json=true","fetch_graph":"https://pith.science/api/pith-number/ASXSVAGJSWJAU6ZFMKTMR5AIJA/graph.json","fetch_events":"https://pith.science/api/pith-number/ASXSVAGJSWJAU6ZFMKTMR5AIJA/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ASXSVAGJSWJAU6ZFMKTMR5AIJA/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ASXSVAGJSWJAU6ZFMKTMR5AIJA/action/storage_attestation","attest_author":"https://pith.science/pith/ASXSVAGJSWJAU6ZFMKTMR5AIJA/action/author_attestation","sign_citation":"https://pith.science/pith/ASXSVAGJSWJAU6ZFMKTMR5AIJA/action/citation_signature","submit_replication":"https://pith.science/pith/ASXSVAGJSWJAU6ZFMKTMR5AIJA/action/replication_record"}},"created_at":"2026-07-05T10:31:53.893495+00:00","updated_at":"2026-07-05T10:31:53.893495+00:00"}