{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:BZW7C257OGL4AZPPVEHNDULVAI","short_pith_number":"pith:BZW7C257","schema_version":"1.0","canonical_sha256":"0e6df16bbf7197c065efa90ed1d1750214ee97bd4d8332c0d5cbfb812d673975","source":{"kind":"arxiv","id":"2406.07177","version":1},"attestation_state":"computed","paper":{"title":"TernaryLLM: Ternarized Large Language Model","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Dong Li, Emad Barsoum, Jian Cheng, Lu Tian, Peisong Wang, Tianqi Chen, Weixiang Xu, Zeyu Zhu, Zhe Li","submitted_at":"2024-06-11T11:40:12Z","abstract_excerpt":"Large language models (LLMs) have achieved remarkable performance on Natural Language Processing (NLP) tasks, but they are hindered by high computational costs and memory requirements. Ternarization, an extreme form of quantization, offers a solution by reducing memory usage and enabling energy-efficient floating-point additions. However, applying ternarization to LLMs faces challenges stemming from outliers in both weights and activations. In this work, observing asymmetric outliers and non-zero means in weights, we introduce Dual Learnable Ternarization (DLT), which enables both scales and s"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2406.07177","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.LG","submitted_at":"2024-06-11T11:40:12Z","cross_cats_sorted":[],"title_canon_sha256":"0a3ac07d16881b8194c27629c22c06fbca5a84fdcef0e1b4ece4d41ebea3a102","abstract_canon_sha256":"11ff7f4a1ebfcf73fbde2971d560efadca9dfe39980b157dfc9e75a73cbe73ae"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:30:19.534185Z","signature_b64":"J2AsbMPgMSUa3T/LLSMDdca7675knUIvpUglIFJrpKfJnUBw4s61L6nDVbiuv+P/p4Tv5DjiDyeRtu66M0SCBA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"0e6df16bbf7197c065efa90ed1d1750214ee97bd4d8332c0d5cbfb812d673975","last_reissued_at":"2026-07-05T08:30:19.533682Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:30:19.533682Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"TernaryLLM: Ternarized Large Language Model","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":[],"primary_cat":"cs.LG","authors_text":"Dong Li, Emad Barsoum, Jian Cheng, Lu Tian, Peisong Wang, Tianqi Chen, Weixiang Xu, Zeyu Zhu, Zhe Li","submitted_at":"2024-06-11T11:40:12Z","abstract_excerpt":"Large language models (LLMs) have achieved remarkable performance on Natural Language Processing (NLP) tasks, but they are hindered by high computational costs and memory requirements. Ternarization, an extreme form of quantization, offers a solution by reducing memory usage and enabling energy-efficient floating-point additions. However, applying ternarization to LLMs faces challenges stemming from outliers in both weights and activations. In this work, observing asymmetric outliers and non-zero means in weights, we introduce Dual Learnable Ternarization (DLT), which enables both scales and s"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2406.07177","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2406.07177/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2406.07177","created_at":"2026-07-05T08:30:19.533750+00:00"},{"alias_kind":"arxiv_version","alias_value":"2406.07177v1","created_at":"2026-07-05T08:30:19.533750+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2406.07177","created_at":"2026-07-05T08:30:19.533750+00:00"},{"alias_kind":"pith_short_12","alias_value":"BZW7C257OGL4","created_at":"2026-07-05T08:30:19.533750+00:00"},{"alias_kind":"pith_short_16","alias_value":"BZW7C257OGL4AZPP","created_at":"2026-07-05T08:30:19.533750+00:00"},{"alias_kind":"pith_short_8","alias_value":"BZW7C257","created_at":"2026-07-05T08:30:19.533750+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.26650","citing_title":"CAT-Q: Cost-efficient and Accurate Ternary Quantization for LLMs","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2606.18114","citing_title":"Ternary Mamba: Grouped Quantization-Aware Training of W1.58A16 State Space Models","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2606.13054","citing_title":"TWLA: Achieving Ternary Weights and Low-Bit Activations for LLMs via Post-Training Quantization","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2510.09696","citing_title":"Vanishing Contributions: A Unified Framework for Smooth and Iterative Model Compression","ref_index":38,"is_internal_anchor":false},{"citing_arxiv_id":"2604.25183","citing_title":"Hardware Generation and Exploration of Lookup Table-Based Accelerators for 1.58-bit LLM Inference","ref_index":9,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/BZW7C257OGL4AZPPVEHNDULVAI","json":"https://pith.science/pith/BZW7C257OGL4AZPPVEHNDULVAI.json","graph_json":"https://pith.science/api/pith-number/BZW7C257OGL4AZPPVEHNDULVAI/graph.json","events_json":"https://pith.science/api/pith-number/BZW7C257OGL4AZPPVEHNDULVAI/events.json","paper":"https://pith.science/paper/BZW7C257"},"agent_actions":{"view_html":"https://pith.science/pith/BZW7C257OGL4AZPPVEHNDULVAI","download_json":"https://pith.science/pith/BZW7C257OGL4AZPPVEHNDULVAI.json","view_paper":"https://pith.science/paper/BZW7C257","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2406.07177&json=true","fetch_graph":"https://pith.science/api/pith-number/BZW7C257OGL4AZPPVEHNDULVAI/graph.json","fetch_events":"https://pith.science/api/pith-number/BZW7C257OGL4AZPPVEHNDULVAI/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/BZW7C257OGL4AZPPVEHNDULVAI/action/timestamp_anchor","attest_storage":"https://pith.science/pith/BZW7C257OGL4AZPPVEHNDULVAI/action/storage_attestation","attest_author":"https://pith.science/pith/BZW7C257OGL4AZPPVEHNDULVAI/action/author_attestation","sign_citation":"https://pith.science/pith/BZW7C257OGL4AZPPVEHNDULVAI/action/citation_signature","submit_replication":"https://pith.science/pith/BZW7C257OGL4AZPPVEHNDULVAI/action/replication_record"}},"created_at":"2026-07-05T08:30:19.533750+00:00","updated_at":"2026-07-05T08:30:19.533750+00:00"}