{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:2U7CKZVZHBJ3MTQQ7COUUAGTVT","short_pith_number":"pith:2U7CKZVZ","schema_version":"1.0","canonical_sha256":"d53e2566b93853b64e10f89d4a00d3accfdcc9c65d8359d88193403b7e0ab77f","source":{"kind":"arxiv","id":"2504.12285","version":2},"attestation_state":"computed","paper":{"title":"BitNet b1.58 2B4T Technical Report","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Furu Wei, Hongyu Wang, Shaohan Huang, Shuming Ma, Ting Song, Xingxing Zhang, Yan Xia, Ying Hu","submitted_at":"2025-04-16T17:51:43Z","abstract_excerpt":"We introduce BitNet b1.58 2B4T, the first open-source, native 1-bit Large Language Model (LLM) at the 2-billion parameter scale. Trained on a corpus of 4 trillion tokens, the model has been rigorously evaluated across benchmarks covering language understanding, mathematical reasoning, coding proficiency, and conversational ability. Our results demonstrate that BitNet b1.58 2B4T achieves performance on par with leading open-weight, full-precision LLMs of similar size, while offering significant advantages in computational efficiency, including substantially reduced memory footprint, energy cons"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2504.12285","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.CL","submitted_at":"2025-04-16T17:51:43Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"09f78800298f6e3975e09ce1fa76804b74af9f817edaf32b8fb2b3d369f42715","abstract_canon_sha256":"e4cf669c276a83a4f83391fb16af4a0397e5adfbffaf14c46e9fe3008b875212"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:53:54.195295Z","signature_b64":"KuPgjddyo6cKrp4x/wYDVBT0U3qn9bVfYJdwgYpWW5J24DX37pLu+KANLdy0QqR5pONxzqBVQxeiEE+i8f+SAA==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"d53e2566b93853b64e10f89d4a00d3accfdcc9c65d8359d88193403b7e0ab77f","last_reissued_at":"2026-07-05T10:53:54.194750Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:53:54.194750Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"BitNet b1.58 2B4T Technical Report","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.CL","authors_text":"Furu Wei, Hongyu Wang, Shaohan Huang, Shuming Ma, Ting Song, Xingxing Zhang, Yan Xia, Ying Hu","submitted_at":"2025-04-16T17:51:43Z","abstract_excerpt":"We introduce BitNet b1.58 2B4T, the first open-source, native 1-bit Large Language Model (LLM) at the 2-billion parameter scale. Trained on a corpus of 4 trillion tokens, the model has been rigorously evaluated across benchmarks covering language understanding, mathematical reasoning, coding proficiency, and conversational ability. Our results demonstrate that BitNet b1.58 2B4T achieves performance on par with leading open-weight, full-precision LLMs of similar size, while offering significant advantages in computational efficiency, including substantially reduced memory footprint, energy cons"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2504.12285","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2504.12285/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2504.12285","created_at":"2026-07-05T10:53:54.194820+00:00"},{"alias_kind":"arxiv_version","alias_value":"2504.12285v2","created_at":"2026-07-05T10:53:54.194820+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2504.12285","created_at":"2026-07-05T10:53:54.194820+00:00"},{"alias_kind":"pith_short_12","alias_value":"2U7CKZVZHBJ3","created_at":"2026-07-05T10:53:54.194820+00:00"},{"alias_kind":"pith_short_16","alias_value":"2U7CKZVZHBJ3MTQQ","created_at":"2026-07-05T10:53:54.194820+00:00"},{"alias_kind":"pith_short_8","alias_value":"2U7CKZVZ","created_at":"2026-07-05T10:53:54.194820+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":13,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.25674","citing_title":"BitNet Text Embeddings","ref_index":40,"is_internal_anchor":false},{"citing_arxiv_id":"2606.22249","citing_title":"On the Expressive Power of Weight Quantization in Large Language Models","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2606.10531","citing_title":"LC-QAT: Data-Efficient 2-Bit QAT for LLMs via Linear-Constrained Vector Quantization","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2606.10531","citing_title":"LC-QAT: Data-Efficient 2-Bit QAT for LLMs via Linear-Constrained Vector Quantization","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2606.05017","citing_title":"GoldenFloat: A Phi-Derived Static-Split Floating-Point Family from GF4 to GF1024 with a Lucas-Exact Integer Identity","ref_index":25,"is_internal_anchor":false},{"citing_arxiv_id":"2606.03026","citing_title":"Spike-Aware C++ INT8 Inference for Sparse Spiking Language Models on Commodity CPUs","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2508.06974","citing_title":"Rethinking 1-bit Optimization Leveraging Pre-trained Large Language Models","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2512.06443","citing_title":"Vec-LUT: Vector Table Lookup for Parallel Ultra-Low-Bit LLM Inference on Edge Devices","ref_index":28,"is_internal_anchor":false},{"citing_arxiv_id":"2604.02501","citing_title":"ECG Foundation Models and Medical LLMs for Agentic Cardiovascular Intelligence at the Edge: A Review and Outlook","ref_index":160,"is_internal_anchor":false},{"citing_arxiv_id":"2604.27396","citing_title":"VitaLLM: A Versatile, Ultra-Compact Ternary LLM Accelerator with Dependency-Aware Scheduling","ref_index":3,"is_internal_anchor":false},{"citing_arxiv_id":"2605.00320","citing_title":"VitaLLM: A Versatile and Tiny Accelerator for Mixed-Precision LLM Inference on Edge Devices","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2604.19167","citing_title":"LBLLM: Lightweight Binarization of Large Language Models via Three-Stage Distillation","ref_index":92,"is_internal_anchor":false},{"citing_arxiv_id":"2605.01910","citing_title":"Stochastic Sparse Attention for Memory-Bound Inference","ref_index":25,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/2U7CKZVZHBJ3MTQQ7COUUAGTVT","json":"https://pith.science/pith/2U7CKZVZHBJ3MTQQ7COUUAGTVT.json","graph_json":"https://pith.science/api/pith-number/2U7CKZVZHBJ3MTQQ7COUUAGTVT/graph.json","events_json":"https://pith.science/api/pith-number/2U7CKZVZHBJ3MTQQ7COUUAGTVT/events.json","paper":"https://pith.science/paper/2U7CKZVZ"},"agent_actions":{"view_html":"https://pith.science/pith/2U7CKZVZHBJ3MTQQ7COUUAGTVT","download_json":"https://pith.science/pith/2U7CKZVZHBJ3MTQQ7COUUAGTVT.json","view_paper":"https://pith.science/paper/2U7CKZVZ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2504.12285&json=true","fetch_graph":"https://pith.science/api/pith-number/2U7CKZVZHBJ3MTQQ7COUUAGTVT/graph.json","fetch_events":"https://pith.science/api/pith-number/2U7CKZVZHBJ3MTQQ7COUUAGTVT/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/2U7CKZVZHBJ3MTQQ7COUUAGTVT/action/timestamp_anchor","attest_storage":"https://pith.science/pith/2U7CKZVZHBJ3MTQQ7COUUAGTVT/action/storage_attestation","attest_author":"https://pith.science/pith/2U7CKZVZHBJ3MTQQ7COUUAGTVT/action/author_attestation","sign_citation":"https://pith.science/pith/2U7CKZVZHBJ3MTQQ7COUUAGTVT/action/citation_signature","submit_replication":"https://pith.science/pith/2U7CKZVZHBJ3MTQQ7COUUAGTVT/action/replication_record"}},"created_at":"2026-07-05T10:53:54.194820+00:00","updated_at":"2026-07-05T10:53:54.194820+00:00"}