{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:ZYUSPJUWBXBYLG6Z7ZLQCM7IYH","short_pith_number":"pith:ZYUSPJUW","schema_version":"1.0","canonical_sha256":"ce2927a6960dc3859bd9fe570133e8c1cf8ec9a0755f5984cf6950beddc43a55","source":{"kind":"arxiv","id":"2508.11269","version":1},"attestation_state":"computed","paper":{"title":"Inference performance evaluation for LLMs on edge devices with a novel benchmarking framework and metric","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.PF","authors_text":"Bin Yu, Cong Tian, Hao Chen, Jialun Cao, Yepang Liu, Zixuan He","submitted_at":"2025-08-15T07:08:46Z","abstract_excerpt":"With the significant success achieved by large language models (LLMs) like LLaMA, edge computing-based LLM inference services for mobile and PC are in high demand for data privacy. However, different edge platforms have different hardware characteristics and the large demand for memory capacity and bandwidth makes it very challenging to deploy and benchmark LLMs on edge devices. In this paper, we introduce a benchmarking tool named ELIB (edge LLM inference benchmarking) to evaluate LLM inference performance of different edge platforms, and propose a novel metric named MBU to indicate the perce"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2508.11269","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.PF","submitted_at":"2025-08-15T07:08:46Z","cross_cats_sorted":[],"title_canon_sha256":"c96a7a4d6c41575558f9bebd99c59e51ef750877c91d4edb68294878a181abab","abstract_canon_sha256":"c7b16a849fd21a632fe2a68460481f7c9f0553a5b6f8f8f476df6db9f086f669"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:54:21.840258Z","signature_b64":"4u7Hs/84Yk/lQ7h5czkEpmyj2iSx0guqj2yjgzcKhn+yHAIDaEqbl+FmoENj5QYFUluU1mQMjaF19dsx531mBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ce2927a6960dc3859bd9fe570133e8c1cf8ec9a0755f5984cf6950beddc43a55","last_reissued_at":"2026-07-05T11:54:21.839782Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:54:21.839782Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Inference performance evaluation for LLMs on edge devices with a novel benchmarking framework and metric","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.PF","authors_text":"Bin Yu, Cong Tian, Hao Chen, Jialun Cao, Yepang Liu, Zixuan He","submitted_at":"2025-08-15T07:08:46Z","abstract_excerpt":"With the significant success achieved by large language models (LLMs) like LLaMA, edge computing-based LLM inference services for mobile and PC are in high demand for data privacy. However, different edge platforms have different hardware characteristics and the large demand for memory capacity and bandwidth makes it very challenging to deploy and benchmark LLMs on edge devices. In this paper, we introduce a benchmarking tool named ELIB (edge LLM inference benchmarking) to evaluate LLM inference performance of different edge platforms, and propose a novel metric named MBU to indicate the perce"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2508.11269","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2508.11269/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2508.11269","created_at":"2026-07-05T11:54:21.839839+00:00"},{"alias_kind":"arxiv_version","alias_value":"2508.11269v1","created_at":"2026-07-05T11:54:21.839839+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2508.11269","created_at":"2026-07-05T11:54:21.839839+00:00"},{"alias_kind":"pith_short_12","alias_value":"ZYUSPJUWBXBY","created_at":"2026-07-05T11:54:21.839839+00:00"},{"alias_kind":"pith_short_16","alias_value":"ZYUSPJUWBXBYLG6Z","created_at":"2026-07-05T11:54:21.839839+00:00"},{"alias_kind":"pith_short_8","alias_value":"ZYUSPJUW","created_at":"2026-07-05T11:54:21.839839+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":2,"sample":[{"citing_arxiv_id":"2607.05876","citing_title":"Think Before You Grid-Search: Floor-First Triage for LLM Serving","ref_index":6,"is_internal_anchor":true},{"citing_arxiv_id":"2607.05876","citing_title":"Think Before You Grid-Search: Floor-First Triage for LLM Serving","ref_index":6,"is_internal_anchor":true},{"citing_arxiv_id":"2605.11186","citing_title":"CATS: Cascaded Adaptive Tree Speculation for Memory-Limited LLM Inference Acceleration","ref_index":47,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ZYUSPJUWBXBYLG6Z7ZLQCM7IYH","json":"https://pith.science/pith/ZYUSPJUWBXBYLG6Z7ZLQCM7IYH.json","graph_json":"https://pith.science/api/pith-number/ZYUSPJUWBXBYLG6Z7ZLQCM7IYH/graph.json","events_json":"https://pith.science/api/pith-number/ZYUSPJUWBXBYLG6Z7ZLQCM7IYH/events.json","paper":"https://pith.science/paper/ZYUSPJUW"},"agent_actions":{"view_html":"https://pith.science/pith/ZYUSPJUWBXBYLG6Z7ZLQCM7IYH","download_json":"https://pith.science/pith/ZYUSPJUWBXBYLG6Z7ZLQCM7IYH.json","view_paper":"https://pith.science/paper/ZYUSPJUW","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2508.11269&json=true","fetch_graph":"https://pith.science/api/pith-number/ZYUSPJUWBXBYLG6Z7ZLQCM7IYH/graph.json","fetch_events":"https://pith.science/api/pith-number/ZYUSPJUWBXBYLG6Z7ZLQCM7IYH/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ZYUSPJUWBXBYLG6Z7ZLQCM7IYH/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ZYUSPJUWBXBYLG6Z7ZLQCM7IYH/action/storage_attestation","attest_author":"https://pith.science/pith/ZYUSPJUWBXBYLG6Z7ZLQCM7IYH/action/author_attestation","sign_citation":"https://pith.science/pith/ZYUSPJUWBXBYLG6Z7ZLQCM7IYH/action/citation_signature","submit_replication":"https://pith.science/pith/ZYUSPJUWBXBYLG6Z7ZLQCM7IYH/action/replication_record"}},"created_at":"2026-07-05T11:54:21.839839+00:00","updated_at":"2026-07-05T11:54:21.839839+00:00"}