{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:IRALRQN7Q45TKUNA6RKNXVIAQM","short_pith_number":"pith:IRALRQN7","schema_version":"1.0","canonical_sha256":"4440b8c1bf873b3551a0f454dbd500831c8875edca447e15692e7cfb4cbc3dcb","source":{"kind":"arxiv","id":"2501.14925","version":2},"attestation_state":"computed","paper":{"title":"Profiling Apple Silicon Performance for ML Training","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.PF","authors_text":"Dahua Feng, Felix Xiaozhu Lin, Rongxiang Wang, Zhiming Xu","submitted_at":"2025-01-24T21:31:11Z","abstract_excerpt":"Apple Silicon has attracted much attention for its performance and role in machine learning (ML) training. Unlike NVIDIA GPUs, which have traditionally dominated ML training, Apple Silicon has a significant difference in memory architecture. It uses Unified Memory, which integrates CPU and GPU memory instead of separate CPU memory and GPU VRAM. However, it is difficult to tell whether Unified Memory means more performance benefits.\n  This paper investigates the performance differences by training several large language model (LLM) workloads end-to-end under different memory scenarios. The resu"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2501.14925","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.PF","submitted_at":"2025-01-24T21:31:11Z","cross_cats_sorted":[],"title_canon_sha256":"2ea296731ad57d70118e695664272063d6acf6679dd72964e791083b2a1213eb","abstract_canon_sha256":"31b75559aa73e699d79b70057401245f1e77d2b4a3b0d4cbc19f7a142af65eec"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:06:33.081535Z","signature_b64":"Y7SY6wqI1bm3MRDH8bsU3aFz8YOdAITqidV4GUa1+xSDx6OStcovfGf8TsSfzRvsLTmMjxCPjn+iFTBCdBApAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"4440b8c1bf873b3551a0f454dbd500831c8875edca447e15692e7cfb4cbc3dcb","last_reissued_at":"2026-07-05T10:06:33.081073Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:06:33.081073Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Profiling Apple Silicon Performance for ML Training","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.PF","authors_text":"Dahua Feng, Felix Xiaozhu Lin, Rongxiang Wang, Zhiming Xu","submitted_at":"2025-01-24T21:31:11Z","abstract_excerpt":"Apple Silicon has attracted much attention for its performance and role in machine learning (ML) training. Unlike NVIDIA GPUs, which have traditionally dominated ML training, Apple Silicon has a significant difference in memory architecture. It uses Unified Memory, which integrates CPU and GPU memory instead of separate CPU memory and GPU VRAM. However, it is difficult to tell whether Unified Memory means more performance benefits.\n  This paper investigates the performance differences by training several large language model (LLM) workloads end-to-end under different memory scenarios. The resu"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2501.14925","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2501.14925/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2501.14925","created_at":"2026-07-05T10:06:33.081135+00:00"},{"alias_kind":"arxiv_version","alias_value":"2501.14925v2","created_at":"2026-07-05T10:06:33.081135+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2501.14925","created_at":"2026-07-05T10:06:33.081135+00:00"},{"alias_kind":"pith_short_12","alias_value":"IRALRQN7Q45T","created_at":"2026-07-05T10:06:33.081135+00:00"},{"alias_kind":"pith_short_16","alias_value":"IRALRQN7Q45TKUNA","created_at":"2026-07-05T10:06:33.081135+00:00"},{"alias_kind":"pith_short_8","alias_value":"IRALRQN7","created_at":"2026-07-05T10:06:33.081135+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":2,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.12765","citing_title":"Rigel: Reverse-Engineering the Metal 4.1 Tensor Compute Path on the Apple M4 Max GPU","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2502.15761","citing_title":"AIvaluateXR: An Evaluation Framework for on-Device AI in XR with Benchmarking Results","ref_index":59,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/IRALRQN7Q45TKUNA6RKNXVIAQM","json":"https://pith.science/pith/IRALRQN7Q45TKUNA6RKNXVIAQM.json","graph_json":"https://pith.science/api/pith-number/IRALRQN7Q45TKUNA6RKNXVIAQM/graph.json","events_json":"https://pith.science/api/pith-number/IRALRQN7Q45TKUNA6RKNXVIAQM/events.json","paper":"https://pith.science/paper/IRALRQN7"},"agent_actions":{"view_html":"https://pith.science/pith/IRALRQN7Q45TKUNA6RKNXVIAQM","download_json":"https://pith.science/pith/IRALRQN7Q45TKUNA6RKNXVIAQM.json","view_paper":"https://pith.science/paper/IRALRQN7","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2501.14925&json=true","fetch_graph":"https://pith.science/api/pith-number/IRALRQN7Q45TKUNA6RKNXVIAQM/graph.json","fetch_events":"https://pith.science/api/pith-number/IRALRQN7Q45TKUNA6RKNXVIAQM/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/IRALRQN7Q45TKUNA6RKNXVIAQM/action/timestamp_anchor","attest_storage":"https://pith.science/pith/IRALRQN7Q45TKUNA6RKNXVIAQM/action/storage_attestation","attest_author":"https://pith.science/pith/IRALRQN7Q45TKUNA6RKNXVIAQM/action/author_attestation","sign_citation":"https://pith.science/pith/IRALRQN7Q45TKUNA6RKNXVIAQM/action/citation_signature","submit_replication":"https://pith.science/pith/IRALRQN7Q45TKUNA6RKNXVIAQM/action/replication_record"}},"created_at":"2026-07-05T10:06:33.081135+00:00","updated_at":"2026-07-05T10:06:33.081135+00:00"}