{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:MCOE3NPDMAJFVJB3G4X47P7CGU","short_pith_number":"pith:MCOE3NPD","schema_version":"1.0","canonical_sha256":"609c4db5e360125aa43b372fcfbfe235355bf6fdd7e5e017f21855babbf744b5","source":{"kind":"arxiv","id":"2309.04255","version":1},"attestation_state":"computed","paper":{"title":"LLMCad: Fast and Scalable On-device Large Language Model Inference","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.NI","authors_text":"Daliang Xu, Mengwei Xu, Shiyun Wei, Wangsong Yin, Xin Jin, Xuanzhe Liu, Ying Zhang","submitted_at":"2023-09-08T10:44:19Z","abstract_excerpt":"Generative tasks, such as text generation and question answering, hold a crucial position in the realm of mobile applications. Due to their sensitivity to privacy concerns, there is a growing demand for their execution directly on mobile devices. Currently, the execution of these generative tasks heavily depends on Large Language Models (LLMs). Nevertheless, the limited memory capacity of these devices presents a formidable challenge to the scalability of such models.\n  In our research, we introduce LLMCad, an innovative on-device inference engine specifically designed for efficient generative"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2309.04255","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.NI","submitted_at":"2023-09-08T10:44:19Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"3431caa8f6e00b2f67c1e75947a631ca206b00336cbd23af2825a4de9a612142","abstract_canon_sha256":"73d9bfcd99f6e951ece1f1389a32d9f29c29844e7c3794f6c42e9a17a1f398d1"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:48:57.303101Z","signature_b64":"MnkMe3sI4a4rWSeTyb9y/ftOLCFvwHI+yKNFpJl3IVWGuLHq4VluvA5OLmERbqYmcn/atCwnvdDc6Lp/d/O9DA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"609c4db5e360125aa43b372fcfbfe235355bf6fdd7e5e017f21855babbf744b5","last_reissued_at":"2026-07-05T06:48:57.302612Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:48:57.302612Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"LLMCad: Fast and Scalable On-device Large Language Model Inference","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.NI","authors_text":"Daliang Xu, Mengwei Xu, Shiyun Wei, Wangsong Yin, Xin Jin, Xuanzhe Liu, Ying Zhang","submitted_at":"2023-09-08T10:44:19Z","abstract_excerpt":"Generative tasks, such as text generation and question answering, hold a crucial position in the realm of mobile applications. Due to their sensitivity to privacy concerns, there is a growing demand for their execution directly on mobile devices. Currently, the execution of these generative tasks heavily depends on Large Language Models (LLMs). Nevertheless, the limited memory capacity of these devices presents a formidable challenge to the scalability of such models.\n  In our research, we introduce LLMCad, an innovative on-device inference engine specifically designed for efficient generative"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2309.04255","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2309.04255/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2309.04255","created_at":"2026-07-05T06:48:57.302668+00:00"},{"alias_kind":"arxiv_version","alias_value":"2309.04255v1","created_at":"2026-07-05T06:48:57.302668+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2309.04255","created_at":"2026-07-05T06:48:57.302668+00:00"},{"alias_kind":"pith_short_12","alias_value":"MCOE3NPDMAJF","created_at":"2026-07-05T06:48:57.302668+00:00"},{"alias_kind":"pith_short_16","alias_value":"MCOE3NPDMAJFVJB3","created_at":"2026-07-05T06:48:57.302668+00:00"},{"alias_kind":"pith_short_8","alias_value":"MCOE3NPD","created_at":"2026-07-05T06:48:57.302668+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.06485","citing_title":"Litespark Inference For CPUs: Ultra-Fast SIMD Framework for Ternary (1.58-bit) Language Models","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06485","citing_title":"Litespark Inference For CPUs: Ultra-Fast SIMD Framework for Ternary (1.58-bit) Language Models","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2604.14403","citing_title":"A Unified Model and Document Representation for On-Device Retrieval-Augmented Generation","ref_index":50,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/MCOE3NPDMAJFVJB3G4X47P7CGU","json":"https://pith.science/pith/MCOE3NPDMAJFVJB3G4X47P7CGU.json","graph_json":"https://pith.science/api/pith-number/MCOE3NPDMAJFVJB3G4X47P7CGU/graph.json","events_json":"https://pith.science/api/pith-number/MCOE3NPDMAJFVJB3G4X47P7CGU/events.json","paper":"https://pith.science/paper/MCOE3NPD"},"agent_actions":{"view_html":"https://pith.science/pith/MCOE3NPDMAJFVJB3G4X47P7CGU","download_json":"https://pith.science/pith/MCOE3NPDMAJFVJB3G4X47P7CGU.json","view_paper":"https://pith.science/paper/MCOE3NPD","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2309.04255&json=true","fetch_graph":"https://pith.science/api/pith-number/MCOE3NPDMAJFVJB3G4X47P7CGU/graph.json","fetch_events":"https://pith.science/api/pith-number/MCOE3NPDMAJFVJB3G4X47P7CGU/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/MCOE3NPDMAJFVJB3G4X47P7CGU/action/timestamp_anchor","attest_storage":"https://pith.science/pith/MCOE3NPDMAJFVJB3G4X47P7CGU/action/storage_attestation","attest_author":"https://pith.science/pith/MCOE3NPDMAJFVJB3G4X47P7CGU/action/author_attestation","sign_citation":"https://pith.science/pith/MCOE3NPDMAJFVJB3G4X47P7CGU/action/citation_signature","submit_replication":"https://pith.science/pith/MCOE3NPDMAJFVJB3G4X47P7CGU/action/replication_record"}},"created_at":"2026-07-05T06:48:57.302668+00:00","updated_at":"2026-07-05T06:48:57.302668+00:00"}