{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:KSIFQSWBKRZ4SMSEG4MAD2CHEG","short_pith_number":"pith:KSIFQSWB","schema_version":"1.0","canonical_sha256":"5490584ac15473c93244371801e847218055b0b67366e33ebb19f27372d6e5b8","source":{"kind":"arxiv","id":"2408.11556","version":2},"attestation_state":"computed","paper":{"title":"Understanding Data Movement in Tightly Coupled Heterogeneous Systems: A Case Study with the Grace Hopper Superchip","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.DC","authors_text":"Giridhar Chukkapalli, Luigi Fusco, Marcin Chrapek, Mikhail Khalilov, Thomas Schulthess, Torsten Hoefler","submitted_at":"2024-08-21T12:07:54Z","abstract_excerpt":"Heterogeneous supercomputers have become the standard in HPC. GPUs in particular have dominated the accelerator landscape, offering unprecedented performance in parallel workloads and unlocking new possibilities in fields like AI and climate modeling. With many workloads becoming memory-bound, improving the communication latency and bandwidth within the system has become a main driver in the development of new architectures. The Grace Hopper Superchip (GH200) is a significant step in the direction of tightly coupled heterogeneous systems, in which all CPUs and GPUs share a unified address spac"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2408.11556","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by-sa/4.0/","primary_cat":"cs.DC","submitted_at":"2024-08-21T12:07:54Z","cross_cats_sorted":[],"title_canon_sha256":"47fd01182f3170d5fb2181aa0749df7384e95a3e4aa11743215ff2b784f578cf","abstract_canon_sha256":"061b7942bfe53faa839869d41cc01444a65934cb271c9e337266d4dbf42bcc5b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:59:12.129970Z","signature_b64":"iZorT3KYVxl9aIRMTUzxpRKshJuGYp8rSxbxzTBKc6paNVWY8wwyuqjd1+bdO+rSglk4ru/9ag5cOazyZpFaBw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"5490584ac15473c93244371801e847218055b0b67366e33ebb19f27372d6e5b8","last_reissued_at":"2026-07-05T08:59:12.129472Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:59:12.129472Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Understanding Data Movement in Tightly Coupled Heterogeneous Systems: A Case Study with the Grace Hopper Superchip","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.DC","authors_text":"Giridhar Chukkapalli, Luigi Fusco, Marcin Chrapek, Mikhail Khalilov, Thomas Schulthess, Torsten Hoefler","submitted_at":"2024-08-21T12:07:54Z","abstract_excerpt":"Heterogeneous supercomputers have become the standard in HPC. GPUs in particular have dominated the accelerator landscape, offering unprecedented performance in parallel workloads and unlocking new possibilities in fields like AI and climate modeling. With many workloads becoming memory-bound, improving the communication latency and bandwidth within the system has become a main driver in the development of new architectures. The Grace Hopper Superchip (GH200) is a significant step in the direction of tightly coupled heterogeneous systems, in which all CPUs and GPUs share a unified address spac"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2408.11556","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2408.11556/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2408.11556","created_at":"2026-07-05T08:59:12.129542+00:00"},{"alias_kind":"arxiv_version","alias_value":"2408.11556v2","created_at":"2026-07-05T08:59:12.129542+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2408.11556","created_at":"2026-07-05T08:59:12.129542+00:00"},{"alias_kind":"pith_short_12","alias_value":"KSIFQSWBKRZ4","created_at":"2026-07-05T08:59:12.129542+00:00"},{"alias_kind":"pith_short_16","alias_value":"KSIFQSWBKRZ4SMSE","created_at":"2026-07-05T08:59:12.129542+00:00"},{"alias_kind":"pith_short_8","alias_value":"KSIFQSWB","created_at":"2026-07-05T08:59:12.129542+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2601.20309","citing_title":"SuperInfer: SLO-Aware Rotary Scheduling and Memory Management for LLM Inference on Superchips","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2605.15957","citing_title":"To GPU or Not to GPU: Vector Search in Relational Engines","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2604.26074","citing_title":"DAK: Direct-Access-Enabled GPU Memory Offloading with Optimal Efficiency for LLM Inference","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2604.12973","citing_title":"An Engineering Journey Training Large Language Models at Scale on Alps: The Apertus Experience","ref_index":29,"is_internal_anchor":false},{"citing_arxiv_id":"2604.19814","citing_title":"Quantum Integrated High-Performance Computing: Foundations, Architectural Elements and Future Directions","ref_index":41,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/KSIFQSWBKRZ4SMSEG4MAD2CHEG","json":"https://pith.science/pith/KSIFQSWBKRZ4SMSEG4MAD2CHEG.json","graph_json":"https://pith.science/api/pith-number/KSIFQSWBKRZ4SMSEG4MAD2CHEG/graph.json","events_json":"https://pith.science/api/pith-number/KSIFQSWBKRZ4SMSEG4MAD2CHEG/events.json","paper":"https://pith.science/paper/KSIFQSWB"},"agent_actions":{"view_html":"https://pith.science/pith/KSIFQSWBKRZ4SMSEG4MAD2CHEG","download_json":"https://pith.science/pith/KSIFQSWBKRZ4SMSEG4MAD2CHEG.json","view_paper":"https://pith.science/paper/KSIFQSWB","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2408.11556&json=true","fetch_graph":"https://pith.science/api/pith-number/KSIFQSWBKRZ4SMSEG4MAD2CHEG/graph.json","fetch_events":"https://pith.science/api/pith-number/KSIFQSWBKRZ4SMSEG4MAD2CHEG/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/KSIFQSWBKRZ4SMSEG4MAD2CHEG/action/timestamp_anchor","attest_storage":"https://pith.science/pith/KSIFQSWBKRZ4SMSEG4MAD2CHEG/action/storage_attestation","attest_author":"https://pith.science/pith/KSIFQSWBKRZ4SMSEG4MAD2CHEG/action/author_attestation","sign_citation":"https://pith.science/pith/KSIFQSWBKRZ4SMSEG4MAD2CHEG/action/citation_signature","submit_replication":"https://pith.science/pith/KSIFQSWBKRZ4SMSEG4MAD2CHEG/action/replication_record"}},"created_at":"2026-07-05T08:59:12.129542+00:00","updated_at":"2026-07-05T08:59:12.129542+00:00"}