{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:6MNX2SHYLKOO4G3QD53OXDWIMI","short_pith_number":"pith:6MNX2SHY","schema_version":"1.0","canonical_sha256":"f31b7d48f85a9cee1b701f76eb8ec8622242b202cf64504b32e7c4d124a865c8","source":{"kind":"arxiv","id":"2404.11788","version":5},"attestation_state":"computed","paper":{"title":"Understanding the Performance Horizon of the Latest ML Workloads with NonGEMM Workloads","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG","cs.PF"],"primary_cat":"cs.AR","authors_text":"Hyoukjun Kwon, Rachid Karami, Sheng-Chun Kao","submitted_at":"2024-04-17T22:44:22Z","abstract_excerpt":"Among ML operators today, GEneralMatrix Multiplication (GEMM)-based operators are known to be key operators that build the main backbone of ML models. As their computational overhead dominates the overall execution time (e.g., 42.8% - 96.6% in our results), GEMM operators have been the prime optimization targets for fast ML inference. This led to advanced GPUs and accelerators available today, which provided significant boost in the GEMM performance compared to CPUs, aligned with the lesson from Amdahl's law. However, accelerating GEMM has significantly shifted the Amdahl's law's landscape for"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2404.11788","kind":"arxiv","version":5},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AR","submitted_at":"2024-04-17T22:44:22Z","cross_cats_sorted":["cs.LG","cs.PF"],"title_canon_sha256":"2d295b3623189ed74a9053c7a05b475994cb0ddb4660ced109f4835fcd7ad0da","abstract_canon_sha256":"db83a81ed78ec6fff6b5f541b02430ad00ba27b4d72245ef1cd2419c4b6fb6d6"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:49:44.027617Z","signature_b64":"4wcsu/DbV0WKbSa2o8lB7CEoE9VXxwtcU0B6Dl4NxZQsj9UDxsCuLD/pNY/xgQabPk6b77JAuvpMTUHE9O5xBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"f31b7d48f85a9cee1b701f76eb8ec8622242b202cf64504b32e7c4d124a865c8","last_reissued_at":"2026-07-05T10:49:44.027086Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:49:44.027086Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Understanding the Performance Horizon of the Latest ML Workloads with NonGEMM Workloads","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.LG","cs.PF"],"primary_cat":"cs.AR","authors_text":"Hyoukjun Kwon, Rachid Karami, Sheng-Chun Kao","submitted_at":"2024-04-17T22:44:22Z","abstract_excerpt":"Among ML operators today, GEneralMatrix Multiplication (GEMM)-based operators are known to be key operators that build the main backbone of ML models. As their computational overhead dominates the overall execution time (e.g., 42.8% - 96.6% in our results), GEMM operators have been the prime optimization targets for fast ML inference. This led to advanced GPUs and accelerators available today, which provided significant boost in the GEMM performance compared to CPUs, aligned with the lesson from Amdahl's law. However, accelerating GEMM has significantly shifted the Amdahl's law's landscape for"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2404.11788","kind":"arxiv","version":5},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2404.11788/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2404.11788","created_at":"2026-07-05T10:49:44.027148+00:00"},{"alias_kind":"arxiv_version","alias_value":"2404.11788v5","created_at":"2026-07-05T10:49:44.027148+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2404.11788","created_at":"2026-07-05T10:49:44.027148+00:00"},{"alias_kind":"pith_short_12","alias_value":"6MNX2SHYLKOO","created_at":"2026-07-05T10:49:44.027148+00:00"},{"alias_kind":"pith_short_16","alias_value":"6MNX2SHYLKOO4G3Q","created_at":"2026-07-05T10:49:44.027148+00:00"},{"alias_kind":"pith_short_8","alias_value":"6MNX2SHY","created_at":"2026-07-05T10:49:44.027148+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2603.26438","citing_title":"A Lightweight High-Throughput Collective-Capable NoC for Large-Scale ML Accelerators","ref_index":6,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/6MNX2SHYLKOO4G3QD53OXDWIMI","json":"https://pith.science/pith/6MNX2SHYLKOO4G3QD53OXDWIMI.json","graph_json":"https://pith.science/api/pith-number/6MNX2SHYLKOO4G3QD53OXDWIMI/graph.json","events_json":"https://pith.science/api/pith-number/6MNX2SHYLKOO4G3QD53OXDWIMI/events.json","paper":"https://pith.science/paper/6MNX2SHY"},"agent_actions":{"view_html":"https://pith.science/pith/6MNX2SHYLKOO4G3QD53OXDWIMI","download_json":"https://pith.science/pith/6MNX2SHYLKOO4G3QD53OXDWIMI.json","view_paper":"https://pith.science/paper/6MNX2SHY","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2404.11788&json=true","fetch_graph":"https://pith.science/api/pith-number/6MNX2SHYLKOO4G3QD53OXDWIMI/graph.json","fetch_events":"https://pith.science/api/pith-number/6MNX2SHYLKOO4G3QD53OXDWIMI/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/6MNX2SHYLKOO4G3QD53OXDWIMI/action/timestamp_anchor","attest_storage":"https://pith.science/pith/6MNX2SHYLKOO4G3QD53OXDWIMI/action/storage_attestation","attest_author":"https://pith.science/pith/6MNX2SHYLKOO4G3QD53OXDWIMI/action/author_attestation","sign_citation":"https://pith.science/pith/6MNX2SHYLKOO4G3QD53OXDWIMI/action/citation_signature","submit_replication":"https://pith.science/pith/6MNX2SHYLKOO4G3QD53OXDWIMI/action/replication_record"}},"created_at":"2026-07-05T10:49:44.027148+00:00","updated_at":"2026-07-05T10:49:44.027148+00:00"}