{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2026:VAON6VSDMYL7UDCSAFEJ3ASAKC","short_pith_number":"pith:VAON6VSD","schema_version":"1.0","canonical_sha256":"a81cdf56436617fa0c5201489d8240508e20c5e8a9204110b48e63939295a8da","source":{"kind":"arxiv","id":"2607.19438","version":1},"attestation_state":"computed","paper":{"title":"BaseRT: Advancing Best-in-Class LLM Inference with Apple M5 Neural Accelerators","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.DC","cs.LG","cs.PF"],"primary_cat":"cs.AR","authors_text":"Fabian Waschkowski, Lukas Wesemann, Prabod Rathnayaka","submitted_at":"2026-07-21T06:42:18Z","abstract_excerpt":"Apple's M5 generation introduces a redesigned GPU architecture in which every core carries a dedicated Neural Accelerator: on-die matrix units exposed through the Metal~4 tensor API. We show that BaseRT, our native Metal inference runtime for large language models on Apple Silicon, exploits these units to push inference throughput on Apple hardware substantially beyond both llama.cpp and MLX. Building on BaseRT's framework-free design, we add a family of hand-written Metal~4 tensor-core kernels (including dense and mixture-of-experts GEMM and flash-attention prefill kernels) that route the com"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2607.19438","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AR","submitted_at":"2026-07-21T06:42:18Z","cross_cats_sorted":["cs.AI","cs.CL","cs.DC","cs.LG","cs.PF"],"title_canon_sha256":"08285bf3178374953fcd88194321e0df009650a3d219007add33f6af6275db00","abstract_canon_sha256":"b0666f8fb1b1eff696e521e915b640bd8d4f3d663457eef8880d9926d4e661e7"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-23T00:23:49.913300Z","signature_b64":"eyW9ljWHAIhiUO4ujYBwpE7i9GWww/YoW+Ac5JqEoVyHejNvlQd+1n8sAz37ULLAIp8E+vY8pQzzcrMHShABCQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"a81cdf56436617fa0c5201489d8240508e20c5e8a9204110b48e63939295a8da","last_reissued_at":"2026-07-23T00:23:49.912455Z","signature_status":"signed_v1","first_computed_at":"2026-07-23T00:23:49.912455Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"BaseRT: Advancing Best-in-Class LLM Inference with Apple M5 Neural Accelerators","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.DC","cs.LG","cs.PF"],"primary_cat":"cs.AR","authors_text":"Fabian Waschkowski, Lukas Wesemann, Prabod Rathnayaka","submitted_at":"2026-07-21T06:42:18Z","abstract_excerpt":"Apple's M5 generation introduces a redesigned GPU architecture in which every core carries a dedicated Neural Accelerator: on-die matrix units exposed through the Metal~4 tensor API. We show that BaseRT, our native Metal inference runtime for large language models on Apple Silicon, exploits these units to push inference throughput on Apple hardware substantially beyond both llama.cpp and MLX. Building on BaseRT's framework-free design, we add a family of hand-written Metal~4 tensor-core kernels (including dense and mixture-of-experts GEMM and flash-attention prefill kernels) that route the com"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2607.19438","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2607.19438/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2607.19438","created_at":"2026-07-23T00:23:49.912901+00:00"},{"alias_kind":"arxiv_version","alias_value":"2607.19438v1","created_at":"2026-07-23T00:23:49.912901+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2607.19438","created_at":"2026-07-23T00:23:49.912901+00:00"},{"alias_kind":"pith_short_12","alias_value":"VAON6VSDMYL7","created_at":"2026-07-23T00:23:49.912901+00:00"},{"alias_kind":"pith_short_16","alias_value":"VAON6VSDMYL7UDCS","created_at":"2026-07-23T00:23:49.912901+00:00"},{"alias_kind":"pith_short_8","alias_value":"VAON6VSD","created_at":"2026-07-23T00:23:49.912901+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/VAON6VSDMYL7UDCSAFEJ3ASAKC","json":"https://pith.science/pith/VAON6VSDMYL7UDCSAFEJ3ASAKC.json","graph_json":"https://pith.science/api/pith-number/VAON6VSDMYL7UDCSAFEJ3ASAKC/graph.json","events_json":"https://pith.science/api/pith-number/VAON6VSDMYL7UDCSAFEJ3ASAKC/events.json","paper":"https://pith.science/paper/VAON6VSD"},"agent_actions":{"view_html":"https://pith.science/pith/VAON6VSDMYL7UDCSAFEJ3ASAKC","download_json":"https://pith.science/pith/VAON6VSDMYL7UDCSAFEJ3ASAKC.json","view_paper":"https://pith.science/paper/VAON6VSD","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2607.19438&json=true","fetch_graph":"https://pith.science/api/pith-number/VAON6VSDMYL7UDCSAFEJ3ASAKC/graph.json","fetch_events":"https://pith.science/api/pith-number/VAON6VSDMYL7UDCSAFEJ3ASAKC/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/VAON6VSDMYL7UDCSAFEJ3ASAKC/action/timestamp_anchor","attest_storage":"https://pith.science/pith/VAON6VSDMYL7UDCSAFEJ3ASAKC/action/storage_attestation","attest_author":"https://pith.science/pith/VAON6VSDMYL7UDCSAFEJ3ASAKC/action/author_attestation","sign_citation":"https://pith.science/pith/VAON6VSDMYL7UDCSAFEJ3ASAKC/action/citation_signature","submit_replication":"https://pith.science/pith/VAON6VSDMYL7UDCSAFEJ3ASAKC/action/replication_record"}},"created_at":"2026-07-23T00:23:49.912901+00:00","updated_at":"2026-07-23T00:23:49.912901+00:00"}