{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:DXTGS63XWUQ4OVCCXX75UKYLKY","short_pith_number":"pith:DXTGS63X","schema_version":"1.0","canonical_sha256":"1de6697b77b521c75442bdffda2b0b563b007d0100af204818850fd94485c8d2","source":{"kind":"arxiv","id":"2505.13840","version":1},"attestation_state":"computed","paper":{"title":"EfficientLLM: Efficiency in Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Haolong Jia, Huichi Zhou, Jianfeng Gao, Keerthiram Murugesan, Lichao Sun, Lifang He, Rong Zhou, Wei Song, Weixiang Sun, Yanfang Ye, Yixin Liu, Yiyang Li, Yue Huang, Yu Wang, Zhengqing Yuan, Zheyuan Zhang","submitted_at":"2025-05-20T02:27:08Z","abstract_excerpt":"Large Language Models (LLMs) have driven significant progress, yet their growing parameter counts and context windows incur prohibitive compute, energy, and monetary costs. We introduce EfficientLLM, a novel benchmark and the first comprehensive empirical study evaluating efficiency techniques for LLMs at scale. Conducted on a production-class cluster (48xGH200, 8xH200 GPUs), our study systematically explores three key axes: (1) architecture pretraining (efficient attention variants: MQA, GQA, MLA, NSA; sparse Mixture-of-Experts (MoE)), (2) fine-tuning (parameter-efficient methods: LoRA, RSLoR"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2505.13840","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2025-05-20T02:27:08Z","cross_cats_sorted":["cs.AI","cs.LG"],"title_canon_sha256":"6c1c9fd9c88bc6280e4750c086f3c94bde9c452ee4cb6ed142d6f35d1bb8f2cb","abstract_canon_sha256":"5aa4d681b3c498685a67d584d4d60fe83e410db1b4fd38bf885a5dd527436e42"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:06:14.154738Z","signature_b64":"s7AVHI4t72nReZeqfqt279nEOa84fPxpcCb3PugncwCgM6gF3ge0egc5r/IqYTRDjwljbQJb/DcQCBoHAJSJBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"1de6697b77b521c75442bdffda2b0b563b007d0100af204818850fd94485c8d2","last_reissued_at":"2026-07-05T11:06:14.153938Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:06:14.153938Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"EfficientLLM: Efficiency in Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.LG"],"primary_cat":"cs.CL","authors_text":"Haolong Jia, Huichi Zhou, Jianfeng Gao, Keerthiram Murugesan, Lichao Sun, Lifang He, Rong Zhou, Wei Song, Weixiang Sun, Yanfang Ye, Yixin Liu, Yiyang Li, Yue Huang, Yu Wang, Zhengqing Yuan, Zheyuan Zhang","submitted_at":"2025-05-20T02:27:08Z","abstract_excerpt":"Large Language Models (LLMs) have driven significant progress, yet their growing parameter counts and context windows incur prohibitive compute, energy, and monetary costs. We introduce EfficientLLM, a novel benchmark and the first comprehensive empirical study evaluating efficiency techniques for LLMs at scale. Conducted on a production-class cluster (48xGH200, 8xH200 GPUs), our study systematically explores three key axes: (1) architecture pretraining (efficient attention variants: MQA, GQA, MLA, NSA; sparse Mixture-of-Experts (MoE)), (2) fine-tuning (parameter-efficient methods: LoRA, RSLoR"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2505.13840","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2505.13840/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2505.13840","created_at":"2026-07-05T11:06:14.154032+00:00"},{"alias_kind":"arxiv_version","alias_value":"2505.13840v1","created_at":"2026-07-05T11:06:14.154032+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2505.13840","created_at":"2026-07-05T11:06:14.154032+00:00"},{"alias_kind":"pith_short_12","alias_value":"DXTGS63XWUQ4","created_at":"2026-07-05T11:06:14.154032+00:00"},{"alias_kind":"pith_short_16","alias_value":"DXTGS63XWUQ4OVCC","created_at":"2026-07-05T11:06:14.154032+00:00"},{"alias_kind":"pith_short_8","alias_value":"DXTGS63X","created_at":"2026-07-05T11:06:14.154032+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2604.05091","citing_title":"MegaTrain: Full Precision Training of 100B+ Parameter Large Language Models on a Single GPU","ref_index":14,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/DXTGS63XWUQ4OVCCXX75UKYLKY","json":"https://pith.science/pith/DXTGS63XWUQ4OVCCXX75UKYLKY.json","graph_json":"https://pith.science/api/pith-number/DXTGS63XWUQ4OVCCXX75UKYLKY/graph.json","events_json":"https://pith.science/api/pith-number/DXTGS63XWUQ4OVCCXX75UKYLKY/events.json","paper":"https://pith.science/paper/DXTGS63X"},"agent_actions":{"view_html":"https://pith.science/pith/DXTGS63XWUQ4OVCCXX75UKYLKY","download_json":"https://pith.science/pith/DXTGS63XWUQ4OVCCXX75UKYLKY.json","view_paper":"https://pith.science/paper/DXTGS63X","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2505.13840&json=true","fetch_graph":"https://pith.science/api/pith-number/DXTGS63XWUQ4OVCCXX75UKYLKY/graph.json","fetch_events":"https://pith.science/api/pith-number/DXTGS63XWUQ4OVCCXX75UKYLKY/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/DXTGS63XWUQ4OVCCXX75UKYLKY/action/timestamp_anchor","attest_storage":"https://pith.science/pith/DXTGS63XWUQ4OVCCXX75UKYLKY/action/storage_attestation","attest_author":"https://pith.science/pith/DXTGS63XWUQ4OVCCXX75UKYLKY/action/author_attestation","sign_citation":"https://pith.science/pith/DXTGS63XWUQ4OVCCXX75UKYLKY/action/citation_signature","submit_replication":"https://pith.science/pith/DXTGS63XWUQ4OVCCXX75UKYLKY/action/replication_record"}},"created_at":"2026-07-05T11:06:14.154032+00:00","updated_at":"2026-07-05T11:06:14.154032+00:00"}