{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:2JM7XEXTSMBPT3C6L7LHUGZ4W5","short_pith_number":"pith:2JM7XEXT","schema_version":"1.0","canonical_sha256":"d259fb92f39302f9ec5e5fd67a1b3cb74dd3ebc54d5993e82fd36c7bc1c1ad9f","source":{"kind":"arxiv","id":"2504.17674","version":1},"attestation_state":"computed","paper":{"title":"Energy Considerations of Large Language Model Inference and Efficiency Optimizations","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Clara Na, Emma Strubell, Jared Fernandez, Sasha Luccioni, Vashisth Tiwari, Yonatan Bisk","submitted_at":"2025-04-24T15:45:05Z","abstract_excerpt":"As large language models (LLMs) scale in size and adoption, their computational and environmental costs continue to rise. Prior benchmarking efforts have primarily focused on latency reduction in idealized settings, often overlooking the diverse real-world inference workloads that shape energy use. In this work, we systematically analyze the energy implications of common inference efficiency optimizations across diverse Natural Language Processing (NLP) and generative Artificial Intelligence (AI) workloads, including conversational AI and code generation. We introduce a modeling approach that "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2504.17674","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-04-24T15:45:05Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"31302dd383bf498ec411fbc71dfad40410d527229d6c9a5f5221f8de163d2486","abstract_canon_sha256":"971e3012831d8bd475725efab00a7c99008097ce983b523b48fb653e89e9b0ca"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:53:33.880008Z","signature_b64":"D5WBd6wiLgNtGsRBU6wMQa3dPj00p1QZKoLIwIEnkVlHl9Xqx6Ph7VPqBhLDRXAbpWp60eXd755bImXaKIFKCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"d259fb92f39302f9ec5e5fd67a1b3cb74dd3ebc54d5993e82fd36c7bc1c1ad9f","last_reissued_at":"2026-07-05T10:53:33.879512Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:53:33.879512Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Energy Considerations of Large Language Model Inference and Efficiency Optimizations","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Clara Na, Emma Strubell, Jared Fernandez, Sasha Luccioni, Vashisth Tiwari, Yonatan Bisk","submitted_at":"2025-04-24T15:45:05Z","abstract_excerpt":"As large language models (LLMs) scale in size and adoption, their computational and environmental costs continue to rise. Prior benchmarking efforts have primarily focused on latency reduction in idealized settings, often overlooking the diverse real-world inference workloads that shape energy use. In this work, we systematically analyze the energy implications of common inference efficiency optimizations across diverse Natural Language Processing (NLP) and generative Artificial Intelligence (AI) workloads, including conversational AI and code generation. We introduce a modeling approach that "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2504.17674","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2504.17674/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2504.17674","created_at":"2026-07-05T10:53:33.879570+00:00"},{"alias_kind":"arxiv_version","alias_value":"2504.17674v1","created_at":"2026-07-05T10:53:33.879570+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2504.17674","created_at":"2026-07-05T10:53:33.879570+00:00"},{"alias_kind":"pith_short_12","alias_value":"2JM7XEXTSMBP","created_at":"2026-07-05T10:53:33.879570+00:00"},{"alias_kind":"pith_short_16","alias_value":"2JM7XEXTSMBPT3C6","created_at":"2026-07-05T10:53:33.879570+00:00"},{"alias_kind":"pith_short_8","alias_value":"2JM7XEXT","created_at":"2026-07-05T10:53:33.879570+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.10861","citing_title":"From Perception to Action: Can UI Interventions Foster Sustainable LLM Chatbot","ref_index":67,"is_internal_anchor":false},{"citing_arxiv_id":"2606.08096","citing_title":"Identifying unique developers in OSS projects: A family of models","ref_index":125,"is_internal_anchor":false},{"citing_arxiv_id":"2607.00274","citing_title":"SEFORA: Student Essays with Feedback Corpus and LLM Feedback Evaluation Framework","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2602.05695","citing_title":"SweetSpot: An Analytical Model for Predicting Energy Efficiency of LLM Inference","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2602.17697","citing_title":"Pimp My LLM: Leveraging Variability Modeling to Tune Inference Hyperparameters","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05416","citing_title":"From Cradle to Cloud: A Life Cycle Review of AI's Environmental Footprint","ref_index":35,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/2JM7XEXTSMBPT3C6L7LHUGZ4W5","json":"https://pith.science/pith/2JM7XEXTSMBPT3C6L7LHUGZ4W5.json","graph_json":"https://pith.science/api/pith-number/2JM7XEXTSMBPT3C6L7LHUGZ4W5/graph.json","events_json":"https://pith.science/api/pith-number/2JM7XEXTSMBPT3C6L7LHUGZ4W5/events.json","paper":"https://pith.science/paper/2JM7XEXT"},"agent_actions":{"view_html":"https://pith.science/pith/2JM7XEXTSMBPT3C6L7LHUGZ4W5","download_json":"https://pith.science/pith/2JM7XEXTSMBPT3C6L7LHUGZ4W5.json","view_paper":"https://pith.science/paper/2JM7XEXT","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2504.17674&json=true","fetch_graph":"https://pith.science/api/pith-number/2JM7XEXTSMBPT3C6L7LHUGZ4W5/graph.json","fetch_events":"https://pith.science/api/pith-number/2JM7XEXTSMBPT3C6L7LHUGZ4W5/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/2JM7XEXTSMBPT3C6L7LHUGZ4W5/action/timestamp_anchor","attest_storage":"https://pith.science/pith/2JM7XEXTSMBPT3C6L7LHUGZ4W5/action/storage_attestation","attest_author":"https://pith.science/pith/2JM7XEXTSMBPT3C6L7LHUGZ4W5/action/author_attestation","sign_citation":"https://pith.science/pith/2JM7XEXTSMBPT3C6L7LHUGZ4W5/action/citation_signature","submit_replication":"https://pith.science/pith/2JM7XEXTSMBPT3C6L7LHUGZ4W5/action/replication_record"}},"created_at":"2026-07-05T10:53:33.879570+00:00","updated_at":"2026-07-05T10:53:33.879570+00:00"}