{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:7KICAGXOXJ4TD2CFRO34JVTPFD","short_pith_number":"pith:7KICAGXO","schema_version":"1.0","canonical_sha256":"fa90201aeeba7931e8458bb7c4d66f28d0350994051f95d9bff236c53faa537c","source":{"kind":"arxiv","id":"2501.02600","version":1},"attestation_state":"computed","paper":{"title":"TAPAS: Thermal- and Power-Aware Scheduling for LLM Inference in Cloud Platforms","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.DC","authors_text":"Chaojie Zhang, Esha Choukse, Haoran Qiu, \\'I\\~nigo Goiri, Josep Torrellas, Jovan Stojkovic, Ricardo Bianchini, Rodrigo Fonseca","submitted_at":"2025-01-05T16:51:17Z","abstract_excerpt":"The rising demand for generative large language models (LLMs) poses challenges for thermal and power management in cloud datacenters. Traditional techniques often are inadequate for LLM inference due to the fine-grained, millisecond-scale execution phases, each with distinct performance, thermal, and power profiles. Additionally, LLM inference workloads are sensitive to various configuration parameters (e.g., model parallelism, size, and quantization) that involve trade-offs between performance, temperature, power, and output quality. Moreover, clouds often co-locate SaaS and IaaS workloads, e"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2501.02600","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.DC","submitted_at":"2025-01-05T16:51:17Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"7883fd0aa0e6ccd8d045ffbc7871f2045f37dfa24446681080781057b7c3c1bf","abstract_canon_sha256":"f0be8a4c940b166b6cca4807598c828968167fd4ce2156cc73ee9c7ba2c3330f"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:57:09.559247Z","signature_b64":"rfiTEC2I/oTqXiy7Z4kb+yCd1R16qsGU7J+U/5VxvY/5HUm1jkh2j8OxfDbms/IPZHt7iRKSlj96UqbmY9gpBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"fa90201aeeba7931e8458bb7c4d66f28d0350994051f95d9bff236c53faa537c","last_reissued_at":"2026-07-05T09:57:09.558851Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:57:09.558851Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"TAPAS: Thermal- and Power-Aware Scheduling for LLM Inference in Cloud Platforms","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.DC","authors_text":"Chaojie Zhang, Esha Choukse, Haoran Qiu, \\'I\\~nigo Goiri, Josep Torrellas, Jovan Stojkovic, Ricardo Bianchini, Rodrigo Fonseca","submitted_at":"2025-01-05T16:51:17Z","abstract_excerpt":"The rising demand for generative large language models (LLMs) poses challenges for thermal and power management in cloud datacenters. Traditional techniques often are inadequate for LLM inference due to the fine-grained, millisecond-scale execution phases, each with distinct performance, thermal, and power profiles. Additionally, LLM inference workloads are sensitive to various configuration parameters (e.g., model parallelism, size, and quantization) that involve trade-offs between performance, temperature, power, and output quality. Moreover, clouds often co-locate SaaS and IaaS workloads, e"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2501.02600","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2501.02600/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2501.02600","created_at":"2026-07-05T09:57:09.558897+00:00"},{"alias_kind":"arxiv_version","alias_value":"2501.02600v1","created_at":"2026-07-05T09:57:09.558897+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2501.02600","created_at":"2026-07-05T09:57:09.558897+00:00"},{"alias_kind":"pith_short_12","alias_value":"7KICAGXOXJ4T","created_at":"2026-07-05T09:57:09.558897+00:00"},{"alias_kind":"pith_short_16","alias_value":"7KICAGXOXJ4TD2CF","created_at":"2026-07-05T09:57:09.558897+00:00"},{"alias_kind":"pith_short_8","alias_value":"7KICAGXO","created_at":"2026-07-05T09:57:09.558897+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2604.09593","citing_title":"Benchmarking Compound AI Applications for Hardware-Software Co-Design","ref_index":28,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/7KICAGXOXJ4TD2CFRO34JVTPFD","json":"https://pith.science/pith/7KICAGXOXJ4TD2CFRO34JVTPFD.json","graph_json":"https://pith.science/api/pith-number/7KICAGXOXJ4TD2CFRO34JVTPFD/graph.json","events_json":"https://pith.science/api/pith-number/7KICAGXOXJ4TD2CFRO34JVTPFD/events.json","paper":"https://pith.science/paper/7KICAGXO"},"agent_actions":{"view_html":"https://pith.science/pith/7KICAGXOXJ4TD2CFRO34JVTPFD","download_json":"https://pith.science/pith/7KICAGXOXJ4TD2CFRO34JVTPFD.json","view_paper":"https://pith.science/paper/7KICAGXO","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2501.02600&json=true","fetch_graph":"https://pith.science/api/pith-number/7KICAGXOXJ4TD2CFRO34JVTPFD/graph.json","fetch_events":"https://pith.science/api/pith-number/7KICAGXOXJ4TD2CFRO34JVTPFD/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/7KICAGXOXJ4TD2CFRO34JVTPFD/action/timestamp_anchor","attest_storage":"https://pith.science/pith/7KICAGXOXJ4TD2CFRO34JVTPFD/action/storage_attestation","attest_author":"https://pith.science/pith/7KICAGXOXJ4TD2CFRO34JVTPFD/action/author_attestation","sign_citation":"https://pith.science/pith/7KICAGXOXJ4TD2CFRO34JVTPFD/action/citation_signature","submit_replication":"https://pith.science/pith/7KICAGXOXJ4TD2CFRO34JVTPFD/action/replication_record"}},"created_at":"2026-07-05T09:57:09.558897+00:00","updated_at":"2026-07-05T09:57:09.558897+00:00"}