{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:INHEMWERZYKVAWA4PAI6I3PH3M","short_pith_number":"pith:INHEMWER","schema_version":"1.0","canonical_sha256":"434e465891ce1550581c7811e46de7db28e44cce6f3084e89aa4af0ff578785d","source":{"kind":"arxiv","id":"2401.14112","version":2},"attestation_state":"computed","paper":{"title":"FP6-LLM: Efficiently Serving Large Language Models Through FP6-Centric Algorithm-System Co-Design","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.AR"],"primary_cat":"cs.LG","authors_text":"Arash Bakhtiari, Donglin Zhuang, Haojun Xia, Michael Wyatt, Olatunji Ruwase, Shiyang Chen, Shuaiwen Leon Song, Stephen Youn, Xiaoxia Wu, Yuxiong He, Zhen Zheng, Zhewei Yao, Zhongzhu Zhou","submitted_at":"2024-01-25T11:46:38Z","abstract_excerpt":"Six-bit quantization (FP6) can effectively reduce the size of large language models (LLMs) and preserve the model quality consistently across varied applications. However, existing systems do not provide Tensor Core support for FP6 quantization and struggle to achieve practical performance improvements during LLM inference. It is challenging to support FP6 quantization on GPUs due to (1) unfriendly memory access of model weights with irregular bit-width and (2) high runtime overhead of weight de-quantization. To address these problems, we propose TC-FPx, the first full-stack GPU kernel design "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2401.14112","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-01-25T11:46:38Z","cross_cats_sorted":["cs.AI","cs.AR"],"title_canon_sha256":"eea3d937d529482bd9966fcf4a8d7f30afb7c720d250cebae63f31ccffa27dbe","abstract_canon_sha256":"fb1c483f20875ff8058ba19a05ee01c4de66a606819df52fc7fa42d9583a12db"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:51:38.516309Z","signature_b64":"yAxVCqx36BakTxyEwDwI5N8+YgyK9Q6q5xNBTibUtWL0ndStF2gvZyubyoJrxQt2dFLrdDwtRPcmyYOit1AwAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"434e465891ce1550581c7811e46de7db28e44cce6f3084e89aa4af0ff578785d","last_reissued_at":"2026-07-05T07:51:38.515816Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:51:38.515816Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"FP6-LLM: Efficiently Serving Large Language Models Through FP6-Centric Algorithm-System Co-Design","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.AR"],"primary_cat":"cs.LG","authors_text":"Arash Bakhtiari, Donglin Zhuang, Haojun Xia, Michael Wyatt, Olatunji Ruwase, Shiyang Chen, Shuaiwen Leon Song, Stephen Youn, Xiaoxia Wu, Yuxiong He, Zhen Zheng, Zhewei Yao, Zhongzhu Zhou","submitted_at":"2024-01-25T11:46:38Z","abstract_excerpt":"Six-bit quantization (FP6) can effectively reduce the size of large language models (LLMs) and preserve the model quality consistently across varied applications. However, existing systems do not provide Tensor Core support for FP6 quantization and struggle to achieve practical performance improvements during LLM inference. It is challenging to support FP6 quantization on GPUs due to (1) unfriendly memory access of model weights with irregular bit-width and (2) high runtime overhead of weight de-quantization. To address these problems, we propose TC-FPx, the first full-stack GPU kernel design "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2401.14112","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2401.14112/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2401.14112","created_at":"2026-07-05T07:51:38.515876+00:00"},{"alias_kind":"arxiv_version","alias_value":"2401.14112v2","created_at":"2026-07-05T07:51:38.515876+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2401.14112","created_at":"2026-07-05T07:51:38.515876+00:00"},{"alias_kind":"pith_short_12","alias_value":"INHEMWERZYKV","created_at":"2026-07-05T07:51:38.515876+00:00"},{"alias_kind":"pith_short_16","alias_value":"INHEMWERZYKVAWA4","created_at":"2026-07-05T07:51:38.515876+00:00"},{"alias_kind":"pith_short_8","alias_value":"INHEMWER","created_at":"2026-07-05T07:51:38.515876+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2502.03589","citing_title":"HACK: Homomorphic Acceleration via Compression of the Key-Value Cache for Disaggregated LLM Inference","ref_index":47,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/INHEMWERZYKVAWA4PAI6I3PH3M","json":"https://pith.science/pith/INHEMWERZYKVAWA4PAI6I3PH3M.json","graph_json":"https://pith.science/api/pith-number/INHEMWERZYKVAWA4PAI6I3PH3M/graph.json","events_json":"https://pith.science/api/pith-number/INHEMWERZYKVAWA4PAI6I3PH3M/events.json","paper":"https://pith.science/paper/INHEMWER"},"agent_actions":{"view_html":"https://pith.science/pith/INHEMWERZYKVAWA4PAI6I3PH3M","download_json":"https://pith.science/pith/INHEMWERZYKVAWA4PAI6I3PH3M.json","view_paper":"https://pith.science/paper/INHEMWER","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2401.14112&json=true","fetch_graph":"https://pith.science/api/pith-number/INHEMWERZYKVAWA4PAI6I3PH3M/graph.json","fetch_events":"https://pith.science/api/pith-number/INHEMWERZYKVAWA4PAI6I3PH3M/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/INHEMWERZYKVAWA4PAI6I3PH3M/action/timestamp_anchor","attest_storage":"https://pith.science/pith/INHEMWERZYKVAWA4PAI6I3PH3M/action/storage_attestation","attest_author":"https://pith.science/pith/INHEMWERZYKVAWA4PAI6I3PH3M/action/author_attestation","sign_citation":"https://pith.science/pith/INHEMWERZYKVAWA4PAI6I3PH3M/action/citation_signature","submit_replication":"https://pith.science/pith/INHEMWERZYKVAWA4PAI6I3PH3M/action/replication_record"}},"created_at":"2026-07-05T07:51:38.515876+00:00","updated_at":"2026-07-05T07:51:38.515876+00:00"}