{"bundle_type":"pith_open_graph_bundle","bundle_version":"1.0","pith_number":"pith:2013:RSCAW5K7WJLHRVWATVPXB2DT7K","short_pith_number":"pith:RSCAW5K7","canonical_record":{"source":{"id":"1304.7053","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.MS","submitted_at":"2013-04-26T02:22:14Z","cross_cats_sorted":["cs.DC","math.NA"],"title_canon_sha256":"6e9a2329779f1441d1effa2a74369c2e5452236c50a23f374600c748d22a0d8c","abstract_canon_sha256":"2ac92c699d168d939806cf81ac7b5db4d25192d4d2a371f86a1d5e076e189ab1"},"schema_version":"1.0"},"canonical_sha256":"8c840b755fb25678d6c09d5f70e873fa9d7a98d5991026f6d15ff7b34dcce60b","source":{"kind":"arxiv","id":"1304.7053","version":1},"source_aliases":[{"alias_kind":"arxiv","alias_value":"1304.7053","created_at":"2026-05-18T03:27:02Z"},{"alias_kind":"arxiv_version","alias_value":"1304.7053v1","created_at":"2026-05-18T03:27:02Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1304.7053","created_at":"2026-05-18T03:27:02Z"},{"alias_kind":"pith_short_12","alias_value":"RSCAW5K7WJLH","created_at":"2026-05-18T12:27:59Z"},{"alias_kind":"pith_short_16","alias_value":"RSCAW5K7WJLHRVWA","created_at":"2026-05-18T12:27:59Z"},{"alias_kind":"pith_short_8","alias_value":"RSCAW5K7","created_at":"2026-05-18T12:27:59Z"}],"events":[{"event_type":"record_created","subject_pith_number":"pith:2013:RSCAW5K7WJLHRVWATVPXB2DT7K","target":"record","payload":{"canonical_record":{"source":{"id":"1304.7053","kind":"arxiv","version":1},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.MS","submitted_at":"2013-04-26T02:22:14Z","cross_cats_sorted":["cs.DC","math.NA"],"title_canon_sha256":"6e9a2329779f1441d1effa2a74369c2e5452236c50a23f374600c748d22a0d8c","abstract_canon_sha256":"2ac92c699d168d939806cf81ac7b5db4d25192d4d2a371f86a1d5e076e189ab1"},"schema_version":"1.0"},"canonical_sha256":"8c840b755fb25678d6c09d5f70e873fa9d7a98d5991026f6d15ff7b34dcce60b","receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-05-18T03:27:02.698384Z","signature_b64":"ZpxFvEZCEhCfMy2U3t4R7kVfn3BMdN7lDtyMAZjIov/RytINDpqiXYl3v3ofgNEOnci6lQTNUzOQMGJLaTDZDQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"8c840b755fb25678d6c09d5f70e873fa9d7a98d5991026f6d15ff7b34dcce60b","last_reissued_at":"2026-05-18T03:27:02.697881Z","signature_status":"signed_v1","first_computed_at":"2026-05-18T03:27:02.697881Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"source_kind":"arxiv","source_id":"1304.7053","source_version":1,"attestation_state":"computed"},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-05-18T03:27:02Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"WnrXBROha8S1S/KVseIwhFn/48Ly7gpGQzxDBUK7Wcf7O711xqkVapqE5swfYyanzsUGeaKv3pc09GCmbWX8DA==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-06-23T05:46:04.754472Z"},"content_sha256":"66e5e974715fe3b2d2ba32a9eb19ed9a083df69a494fb0b30bbcb873cb9ab42e","schema_version":"1.0","event_id":"sha256:66e5e974715fe3b2d2ba32a9eb19ed9a083df69a494fb0b30bbcb873cb9ab42e"},{"event_type":"graph_snapshot","subject_pith_number":"pith:2013:RSCAW5K7WJLHRVWATVPXB2DT7K","target":"graph","payload":{"graph_snapshot":{"paper":{"title":"A GEMM interface and implementation on NVIDIA GPUs for multiple small matrices","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.DC","math.NA"],"primary_cat":"cs.MS","authors_text":"Chetan Jhurani, Paul Mullowney","submitted_at":"2013-04-26T02:22:14Z","abstract_excerpt":"We present an interface and an implementation of the General Matrix Multiply (GEMM) routine for multiple small matrices processed simultaneously on NVIDIA graphics processing units (GPUs). We focus on matrix sizes under 16. The implementation can be easily extended to larger sizes. For single precision matrices, our implementation is 30% to 600% faster than the batched cuBLAS implementation distributed in the CUDA Toolkit 5.0 on NVIDIA Tesla K20c. For example, we obtain 104 GFlop/s and 216 GFlop/s when multiplying 100,000 independent matrix pairs of size 10 and 16, respectively. Similar improv"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1304.7053","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"verdict_id":null},"signer":{"signer_id":"pith.science","signer_type":"pith_registry","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"created_at":"2026-05-18T03:27:02Z","supersedes":[],"prev_event":null,"signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"0zyyYhn0aV6dwrf1qi3Usjosn1Trk/va3mb6D0VcItkzsJOgk3cSPIS+Us21sUK2GxLKfNrRcUhRlSZMRk06BA==","signed_message":"open_graph_event_sha256_bytes","signed_at":"2026-06-23T05:46:04.754807Z"},"content_sha256":"cc4a81578797e57470b818c667bae499a090e4c11ef7221d79e57fbbd734588c","schema_version":"1.0","event_id":"sha256:cc4a81578797e57470b818c667bae499a090e4c11ef7221d79e57fbbd734588c"}],"timestamp_proofs":[],"mirror_hints":[{"mirror_type":"https","name":"Pith Resolver","base_url":"https://pith.science","bundle_url":"https://pith.science/pith/RSCAW5K7WJLHRVWATVPXB2DT7K/bundle.json","state_url":"https://pith.science/pith/RSCAW5K7WJLHRVWATVPXB2DT7K/state.json","well_known_bundle_url":"https://pith.science/.well-known/pith/RSCAW5K7WJLHRVWATVPXB2DT7K/bundle.json","status":"primary"}],"public_keys":[{"key_id":"pith-v1-2026-05","algorithm":"ed25519","format":"raw","public_key_b64":"stVStoiQhXFxp4s2pdzPNoqVNBMojDU/fJ2db5S3CbM=","public_key_hex":"b2d552b68890857171a78b36a5dccf368a953413288c353f7c9d9d6f94b709b3","fingerprint_sha256_b32_first128bits":"RVFV5Z2OI2J3ZUO7ERDEBCYNKS","fingerprint_sha256_hex":"8d4b5ee74e4693bcd1df2446408b0d54","rotates_at":null,"url":"https://pith.science/pith-signing-key.json","notes":"Pith uses this Ed25519 key to sign canonical record SHA-256 digests. Verify with: ed25519_verify(public_key, message=canonical_sha256_bytes, signature=base64decode(signature_b64))."}],"merge_version":"pith-open-graph-merge-v1","built_at":"2026-06-23T05:46:04Z","links":{"resolver":"https://pith.science/pith/RSCAW5K7WJLHRVWATVPXB2DT7K","bundle":"https://pith.science/pith/RSCAW5K7WJLHRVWATVPXB2DT7K/bundle.json","state":"https://pith.science/pith/RSCAW5K7WJLHRVWATVPXB2DT7K/state.json","well_known_bundle":"https://pith.science/.well-known/pith/RSCAW5K7WJLHRVWATVPXB2DT7K/bundle.json"},"state":{"state_type":"pith_open_graph_state","state_version":"1.0","pith_number":"pith:2013:RSCAW5K7WJLHRVWATVPXB2DT7K","merge_version":"pith-open-graph-merge-v1","event_count":2,"valid_event_count":2,"invalid_event_count":0,"equivocation_count":0,"current":{"canonical_record":{"metadata":{"abstract_canon_sha256":"2ac92c699d168d939806cf81ac7b5db4d25192d4d2a371f86a1d5e076e189ab1","cross_cats_sorted":["cs.DC","math.NA"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.MS","submitted_at":"2013-04-26T02:22:14Z","title_canon_sha256":"6e9a2329779f1441d1effa2a74369c2e5452236c50a23f374600c748d22a0d8c"},"schema_version":"1.0","source":{"id":"1304.7053","kind":"arxiv","version":1}},"source_aliases":[{"alias_kind":"arxiv","alias_value":"1304.7053","created_at":"2026-05-18T03:27:02Z"},{"alias_kind":"arxiv_version","alias_value":"1304.7053v1","created_at":"2026-05-18T03:27:02Z"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.1304.7053","created_at":"2026-05-18T03:27:02Z"},{"alias_kind":"pith_short_12","alias_value":"RSCAW5K7WJLH","created_at":"2026-05-18T12:27:59Z"},{"alias_kind":"pith_short_16","alias_value":"RSCAW5K7WJLHRVWA","created_at":"2026-05-18T12:27:59Z"},{"alias_kind":"pith_short_8","alias_value":"RSCAW5K7","created_at":"2026-05-18T12:27:59Z"}],"graph_snapshots":[{"event_id":"sha256:cc4a81578797e57470b818c667bae499a090e4c11ef7221d79e57fbbd734588c","target":"graph","created_at":"2026-05-18T03:27:02Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"graph_snapshot":{"author_claims":{"count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","strong_count":0},"builder_version":"pith-number-builder-2026-05-17-v1","claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"paper":{"abstract_excerpt":"We present an interface and an implementation of the General Matrix Multiply (GEMM) routine for multiple small matrices processed simultaneously on NVIDIA graphics processing units (GPUs). We focus on matrix sizes under 16. The implementation can be easily extended to larger sizes. For single precision matrices, our implementation is 30% to 600% faster than the batched cuBLAS implementation distributed in the CUDA Toolkit 5.0 on NVIDIA Tesla K20c. For example, we obtain 104 GFlop/s and 216 GFlop/s when multiplying 100,000 independent matrix pairs of size 10 and 16, respectively. Similar improv","authors_text":"Chetan Jhurani, Paul Mullowney","cross_cats":["cs.DC","math.NA"],"headline":"","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.MS","submitted_at":"2013-04-26T02:22:14Z","title":"A GEMM interface and implementation on NVIDIA GPUs for multiple small matrices"},"references":{"count":0,"internal_anchors":0,"resolved_work":0,"sample":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"1304.7053","kind":"arxiv","version":1},"verdict":{"created_at":null,"id":null,"model_set":{},"one_line_summary":"","pipeline_version":null,"pith_extraction_headline":"","strongest_claim":"","weakest_assumption":""}},"verdict_id":null}}],"author_attestations":[],"timestamp_anchors":[],"storage_attestations":[],"citation_signatures":[],"replication_records":[],"corrections":[],"mirror_hints":[],"record_created":{"event_id":"sha256:66e5e974715fe3b2d2ba32a9eb19ed9a083df69a494fb0b30bbcb873cb9ab42e","target":"record","created_at":"2026-05-18T03:27:02Z","signer":{"key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signer_id":"pith.science","signer_type":"pith_registry"},"payload":{"attestation_state":"computed","canonical_record":{"metadata":{"abstract_canon_sha256":"2ac92c699d168d939806cf81ac7b5db4d25192d4d2a371f86a1d5e076e189ab1","cross_cats_sorted":["cs.DC","math.NA"],"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.MS","submitted_at":"2013-04-26T02:22:14Z","title_canon_sha256":"6e9a2329779f1441d1effa2a74369c2e5452236c50a23f374600c748d22a0d8c"},"schema_version":"1.0","source":{"id":"1304.7053","kind":"arxiv","version":1}},"canonical_sha256":"8c840b755fb25678d6c09d5f70e873fa9d7a98d5991026f6d15ff7b34dcce60b","receipt":{"algorithm":"ed25519","builder_version":"pith-number-builder-2026-05-17-v1","canonical_sha256":"8c840b755fb25678d6c09d5f70e873fa9d7a98d5991026f6d15ff7b34dcce60b","first_computed_at":"2026-05-18T03:27:02.697881Z","key_id":"pith-v1-2026-05","kind":"pith_receipt","last_reissued_at":"2026-05-18T03:27:02.697881Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","receipt_version":"0.3","signature_b64":"ZpxFvEZCEhCfMy2U3t4R7kVfn3BMdN7lDtyMAZjIov/RytINDpqiXYl3v3ofgNEOnci6lQTNUzOQMGJLaTDZDQ==","signature_status":"signed_v1","signed_at":"2026-05-18T03:27:02.698384Z","signed_message":"canonical_sha256_bytes"},"source_id":"1304.7053","source_kind":"arxiv","source_version":1}}},"equivocations":[],"invalid_events":[],"applied_event_ids":["sha256:66e5e974715fe3b2d2ba32a9eb19ed9a083df69a494fb0b30bbcb873cb9ab42e","sha256:cc4a81578797e57470b818c667bae499a090e4c11ef7221d79e57fbbd734588c"],"state_sha256":"af5d830dca529d72813d0aad52e1c6da2b8202189004d48e1a536ec4a2de9a13"},"bundle_signature":{"signature_status":"signed_v1","algorithm":"ed25519","key_id":"pith-v1-2026-05","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54","signature_b64":"X4galWhTGWNnH32NZnoYdzBj2E9G1UrBqIRvpuW2ed++diUC2LQQCegWJ4lFRngs/ez1fnzNPzqeJQ9vd9MMCQ==","signed_message":"bundle_sha256_bytes","signed_at":"2026-06-23T05:46:04.756722Z","bundle_sha256":"e48eeefc4365507df321bd30fa92e99d4f1c3f55063dc833f34ddc56919a8413"}}