{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:ELK7H5TAMOMU6CSOH23533F233","short_pith_number":"pith:ELK7H5TA","schema_version":"1.0","canonical_sha256":"22d5f3f66063994f0a4e3eb7ddecbadefb7eb86cdcbe58b4eabe2ab7d703b2b3","source":{"kind":"arxiv","id":"2508.00904","version":1},"attestation_state":"computed","paper":{"title":"Forecasting LLM Inference Performance via Hardware-Agnostic Analytical Modeling","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.AR","cs.LG"],"primary_cat":"cs.PF","authors_text":"Ashish Sirasao, Devleena Das, Rajeev Patwari","submitted_at":"2025-07-29T03:08:31Z","abstract_excerpt":"Large language models (LLMs) have been increasingly deployed as local agents on personal devices with CPUs, NPUs and integrated GPUs. However, forecasting inference performance on devices with such heterogeneity remains challenging due to the dynamic compute and memory demands. Existing approaches rely on GPU benchmarking or machine learning-based latency predictors, which are often hardware-specific and lack generalizability. To this end, we introduce LIFE, a lightweight and modular analytical framework that is comprised of modular analytical model of operators, configurable to characterize L"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2508.00904","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.PF","submitted_at":"2025-07-29T03:08:31Z","cross_cats_sorted":["cs.AI","cs.AR","cs.LG"],"title_canon_sha256":"2b3fe2844a80be6b179ccfd88e51dd8a65fb0eb571012ce3b62f48278b49be53","abstract_canon_sha256":"30891391db8348515457656ac93b2fc0906812bee2c15f1564693284681d7064"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:47:29.532630Z","signature_b64":"pkHYztGZBxEQYy9XyJHy50kGcRTJ00scUsgiRpU4rJhAxdgE3mNarU0HmTE0H3WBYv7J5RLh7nN6v3kgL2ZCCA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"22d5f3f66063994f0a4e3eb7ddecbadefb7eb86cdcbe58b4eabe2ab7d703b2b3","last_reissued_at":"2026-07-05T11:47:29.532216Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:47:29.532216Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Forecasting LLM Inference Performance via Hardware-Agnostic Analytical Modeling","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.AR","cs.LG"],"primary_cat":"cs.PF","authors_text":"Ashish Sirasao, Devleena Das, Rajeev Patwari","submitted_at":"2025-07-29T03:08:31Z","abstract_excerpt":"Large language models (LLMs) have been increasingly deployed as local agents on personal devices with CPUs, NPUs and integrated GPUs. However, forecasting inference performance on devices with such heterogeneity remains challenging due to the dynamic compute and memory demands. Existing approaches rely on GPU benchmarking or machine learning-based latency predictors, which are often hardware-specific and lack generalizability. To this end, we introduce LIFE, a lightweight and modular analytical framework that is comprised of modular analytical model of operators, configurable to characterize L"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2508.00904","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2508.00904/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2508.00904","created_at":"2026-07-05T11:47:29.532270+00:00"},{"alias_kind":"arxiv_version","alias_value":"2508.00904v1","created_at":"2026-07-05T11:47:29.532270+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2508.00904","created_at":"2026-07-05T11:47:29.532270+00:00"},{"alias_kind":"pith_short_12","alias_value":"ELK7H5TAMOMU","created_at":"2026-07-05T11:47:29.532270+00:00"},{"alias_kind":"pith_short_16","alias_value":"ELK7H5TAMOMU6CSO","created_at":"2026-07-05T11:47:29.532270+00:00"},{"alias_kind":"pith_short_8","alias_value":"ELK7H5TA","created_at":"2026-07-05T11:47:29.532270+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.18042","citing_title":"Latency Prediction for LLM Inference on NPU Systems","ref_index":32,"is_internal_anchor":false},{"citing_arxiv_id":"2607.02391","citing_title":"WattGPU: Predicting Inference Power and Latency on Unseen GPUs and LLMs","ref_index":16,"is_internal_anchor":false},{"citing_arxiv_id":"2606.04238","citing_title":"Recover-LoRA for Aggressive Quantization: Reclaiming Accuracy in 2-Bit Language Models via Low-Rank Adaptation with Knowledge Distillation on Synthetic Data","ref_index":51,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/ELK7H5TAMOMU6CSOH23533F233","json":"https://pith.science/pith/ELK7H5TAMOMU6CSOH23533F233.json","graph_json":"https://pith.science/api/pith-number/ELK7H5TAMOMU6CSOH23533F233/graph.json","events_json":"https://pith.science/api/pith-number/ELK7H5TAMOMU6CSOH23533F233/events.json","paper":"https://pith.science/paper/ELK7H5TA"},"agent_actions":{"view_html":"https://pith.science/pith/ELK7H5TAMOMU6CSOH23533F233","download_json":"https://pith.science/pith/ELK7H5TAMOMU6CSOH23533F233.json","view_paper":"https://pith.science/paper/ELK7H5TA","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2508.00904&json=true","fetch_graph":"https://pith.science/api/pith-number/ELK7H5TAMOMU6CSOH23533F233/graph.json","fetch_events":"https://pith.science/api/pith-number/ELK7H5TAMOMU6CSOH23533F233/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/ELK7H5TAMOMU6CSOH23533F233/action/timestamp_anchor","attest_storage":"https://pith.science/pith/ELK7H5TAMOMU6CSOH23533F233/action/storage_attestation","attest_author":"https://pith.science/pith/ELK7H5TAMOMU6CSOH23533F233/action/author_attestation","sign_citation":"https://pith.science/pith/ELK7H5TAMOMU6CSOH23533F233/action/citation_signature","submit_replication":"https://pith.science/pith/ELK7H5TAMOMU6CSOH23533F233/action/replication_record"}},"created_at":"2026-07-05T11:47:29.532270+00:00","updated_at":"2026-07-05T11:47:29.532270+00:00"}