{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:GHCRGA6N7NPCTPWH35FFKXTZ7G","short_pith_number":"pith:GHCRGA6N","schema_version":"1.0","canonical_sha256":"31c51303cdfb5e29bec7df4a555e79f9ad8d5fff243172fc1e1d24cde3e3d9bf","source":{"kind":"arxiv","id":"2407.00088","version":2},"attestation_state":"computed","paper":{"title":"T-MAC: CPU Renaissance via Table Lookup for Low-Bit LLM Deployment on Edge","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.DC","authors_text":"Jianyu Wei, Lei Wang, Lingxiao Ma, Mao Yang, Shijie Cao, Ting Cao, Yanyong Zhang","submitted_at":"2024-06-25T08:38:38Z","abstract_excerpt":"The deployment of Large Language Models (LLMs) on edge devices is increasingly important to enhance on-device intelligence. Weight quantization is crucial for reducing the memory footprint of LLMs on devices. However, low-bit LLMs necessitate mixed precision matrix multiplication (mpGEMM) of low precision weights and high precision activations during inference. Existing systems, lacking native support for mpGEMM, resort to dequantize weights for high precision computation. Such an indirect way can lead to a significant inference overhead.\n  In this paper, we introduce T-MAC, an innovative look"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2407.00088","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.DC","submitted_at":"2024-06-25T08:38:38Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"36d4204e1cb7b70878e2ae15d5a6dc31a3da08ab822af45955897ecabf00f7ae","abstract_canon_sha256":"de5a896e30db52ccd46d185e31bb2aede6283c24d0b225329275ade2b46884aa"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:38:34.493387Z","signature_b64":"8ZqpdQifNnY9UHxkkTnAX6DbEgzl3mWRDnGeIpb4XV1lr2fRAmJA55cyCSES0PherKRMaP9SIXSHczosZwLiDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"31c51303cdfb5e29bec7df4a555e79f9ad8d5fff243172fc1e1d24cde3e3d9bf","last_reissued_at":"2026-07-05T10:38:34.492881Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:38:34.492881Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"T-MAC: CPU Renaissance via Table Lookup for Low-Bit LLM Deployment on Edge","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.DC","authors_text":"Jianyu Wei, Lei Wang, Lingxiao Ma, Mao Yang, Shijie Cao, Ting Cao, Yanyong Zhang","submitted_at":"2024-06-25T08:38:38Z","abstract_excerpt":"The deployment of Large Language Models (LLMs) on edge devices is increasingly important to enhance on-device intelligence. Weight quantization is crucial for reducing the memory footprint of LLMs on devices. However, low-bit LLMs necessitate mixed precision matrix multiplication (mpGEMM) of low precision weights and high precision activations during inference. Existing systems, lacking native support for mpGEMM, resort to dequantize weights for high precision computation. Such an indirect way can lead to a significant inference overhead.\n  In this paper, we introduce T-MAC, an innovative look"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2407.00088","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2407.00088/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2407.00088","created_at":"2026-07-05T10:38:34.492939+00:00"},{"alias_kind":"arxiv_version","alias_value":"2407.00088v2","created_at":"2026-07-05T10:38:34.492939+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2407.00088","created_at":"2026-07-05T10:38:34.492939+00:00"},{"alias_kind":"pith_short_12","alias_value":"GHCRGA6N7NPC","created_at":"2026-07-05T10:38:34.492939+00:00"},{"alias_kind":"pith_short_16","alias_value":"GHCRGA6N7NPCTPWH","created_at":"2026-07-05T10:38:34.492939+00:00"},{"alias_kind":"pith_short_8","alias_value":"GHCRGA6N","created_at":"2026-07-05T10:38:34.492939+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.06485","citing_title":"Litespark Inference For CPUs: Ultra-Fast SIMD Framework for Ternary (1.58-bit) Language Models","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2510.05109","citing_title":"Tiny but Mighty: A Software-Hardware Co-Design Approach for Efficient Multimodal Inference on Battery-Powered Small Devices","ref_index":17,"is_internal_anchor":false},{"citing_arxiv_id":"2512.06443","citing_title":"Vec-LUT: Vector Table Lookup for Parallel Ultra-Low-Bit LLM Inference on Edge Devices","ref_index":43,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06485","citing_title":"Litespark Inference For CPUs: Ultra-Fast SIMD Framework for Ternary (1.58-bit) Language Models","ref_index":21,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/GHCRGA6N7NPCTPWH35FFKXTZ7G","json":"https://pith.science/pith/GHCRGA6N7NPCTPWH35FFKXTZ7G.json","graph_json":"https://pith.science/api/pith-number/GHCRGA6N7NPCTPWH35FFKXTZ7G/graph.json","events_json":"https://pith.science/api/pith-number/GHCRGA6N7NPCTPWH35FFKXTZ7G/events.json","paper":"https://pith.science/paper/GHCRGA6N"},"agent_actions":{"view_html":"https://pith.science/pith/GHCRGA6N7NPCTPWH35FFKXTZ7G","download_json":"https://pith.science/pith/GHCRGA6N7NPCTPWH35FFKXTZ7G.json","view_paper":"https://pith.science/paper/GHCRGA6N","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2407.00088&json=true","fetch_graph":"https://pith.science/api/pith-number/GHCRGA6N7NPCTPWH35FFKXTZ7G/graph.json","fetch_events":"https://pith.science/api/pith-number/GHCRGA6N7NPCTPWH35FFKXTZ7G/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/GHCRGA6N7NPCTPWH35FFKXTZ7G/action/timestamp_anchor","attest_storage":"https://pith.science/pith/GHCRGA6N7NPCTPWH35FFKXTZ7G/action/storage_attestation","attest_author":"https://pith.science/pith/GHCRGA6N7NPCTPWH35FFKXTZ7G/action/author_attestation","sign_citation":"https://pith.science/pith/GHCRGA6N7NPCTPWH35FFKXTZ7G/action/citation_signature","submit_replication":"https://pith.science/pith/GHCRGA6N7NPCTPWH35FFKXTZ7G/action/replication_record"}},"created_at":"2026-07-05T10:38:34.492939+00:00","updated_at":"2026-07-05T10:38:34.492939+00:00"}