{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:WRANBW74FCZBHEY6TDSK7JI4HV","short_pith_number":"pith:WRANBW74","schema_version":"1.0","canonical_sha256":"b440d0dbfc28b213931e98e4afa51c3d5fb23fc693983e09485744b8e673f9cb","source":{"kind":"arxiv","id":"2405.17322","version":1},"attestation_state":"computed","paper":{"title":"Evaluation of computational and energy performance in matrix multiplication algorithms on CPU and GPU using MKL, cuBLAS and SYCL","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AR"],"primary_cat":"cs.DC","authors_text":"Carlos J. Barrios H, L.A. Torres, Yves Denneulin","submitted_at":"2024-05-27T16:21:41Z","abstract_excerpt":"Matrix multiplication is fundamental in the backpropagation algorithm used to train deep neural network models. Libraries like Intel's MKL or NVIDIA's cuBLAS implemented new and optimized matrix multiplication techniques that increase performance and reduce computational costs. These techniques can also be implemented in CUDA and SYCL and functions with AVX2 and AVX512 instructions, which have lower performance but better precision. The study compares execution times and power consumption using PAPI and PERF and compares accuracy for different matrix sizes. Comparisons were made on architectur"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2405.17322","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.DC","submitted_at":"2024-05-27T16:21:41Z","cross_cats_sorted":["cs.AR"],"title_canon_sha256":"712595a538be4ad48dc5d167860bb73d05ac7453321dd285bc96f180248cdaaf","abstract_canon_sha256":"d75dfdc93eaf3e86c7f3ec5dd23bdeddd028a87604d4a0818b9d6bf45820ac33"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:23:42.891197Z","signature_b64":"K86UvfBU6ic6O73FF+jmFiAk9i3fCfmKoJG/fKsVubfVyVh16+sjDj5jnWtYsMFQLb79APGuF1m036uYYyZPCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"b440d0dbfc28b213931e98e4afa51c3d5fb23fc693983e09485744b8e673f9cb","last_reissued_at":"2026-07-05T08:23:42.890648Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:23:42.890648Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Evaluation of computational and energy performance in matrix multiplication algorithms on CPU and GPU using MKL, cuBLAS and SYCL","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AR"],"primary_cat":"cs.DC","authors_text":"Carlos J. Barrios H, L.A. Torres, Yves Denneulin","submitted_at":"2024-05-27T16:21:41Z","abstract_excerpt":"Matrix multiplication is fundamental in the backpropagation algorithm used to train deep neural network models. Libraries like Intel's MKL or NVIDIA's cuBLAS implemented new and optimized matrix multiplication techniques that increase performance and reduce computational costs. These techniques can also be implemented in CUDA and SYCL and functions with AVX2 and AVX512 instructions, which have lower performance but better precision. The study compares execution times and power consumption using PAPI and PERF and compares accuracy for different matrix sizes. Comparisons were made on architectur"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2405.17322","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2405.17322/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2405.17322","created_at":"2026-07-05T08:23:42.890712+00:00"},{"alias_kind":"arxiv_version","alias_value":"2405.17322v1","created_at":"2026-07-05T08:23:42.890712+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2405.17322","created_at":"2026-07-05T08:23:42.890712+00:00"},{"alias_kind":"pith_short_12","alias_value":"WRANBW74FCZB","created_at":"2026-07-05T08:23:42.890712+00:00"},{"alias_kind":"pith_short_16","alias_value":"WRANBW74FCZBHEY6","created_at":"2026-07-05T08:23:42.890712+00:00"},{"alias_kind":"pith_short_8","alias_value":"WRANBW74","created_at":"2026-07-05T08:23:42.890712+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.01174","citing_title":"MAC Performance and Algorithmic Optimization in Matrix Multiplication Workloads","ref_index":9,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/WRANBW74FCZBHEY6TDSK7JI4HV","json":"https://pith.science/pith/WRANBW74FCZBHEY6TDSK7JI4HV.json","graph_json":"https://pith.science/api/pith-number/WRANBW74FCZBHEY6TDSK7JI4HV/graph.json","events_json":"https://pith.science/api/pith-number/WRANBW74FCZBHEY6TDSK7JI4HV/events.json","paper":"https://pith.science/paper/WRANBW74"},"agent_actions":{"view_html":"https://pith.science/pith/WRANBW74FCZBHEY6TDSK7JI4HV","download_json":"https://pith.science/pith/WRANBW74FCZBHEY6TDSK7JI4HV.json","view_paper":"https://pith.science/paper/WRANBW74","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2405.17322&json=true","fetch_graph":"https://pith.science/api/pith-number/WRANBW74FCZBHEY6TDSK7JI4HV/graph.json","fetch_events":"https://pith.science/api/pith-number/WRANBW74FCZBHEY6TDSK7JI4HV/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/WRANBW74FCZBHEY6TDSK7JI4HV/action/timestamp_anchor","attest_storage":"https://pith.science/pith/WRANBW74FCZBHEY6TDSK7JI4HV/action/storage_attestation","attest_author":"https://pith.science/pith/WRANBW74FCZBHEY6TDSK7JI4HV/action/author_attestation","sign_citation":"https://pith.science/pith/WRANBW74FCZBHEY6TDSK7JI4HV/action/citation_signature","submit_replication":"https://pith.science/pith/WRANBW74FCZBHEY6TDSK7JI4HV/action/replication_record"}},"created_at":"2026-07-05T08:23:42.890712+00:00","updated_at":"2026-07-05T08:23:42.890712+00:00"}