{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:Y7RMARKNK5TPSAVWDRPLJNRCSG","short_pith_number":"pith:Y7RMARKN","schema_version":"1.0","canonical_sha256":"c7e2c0454d5766f902b61c5eb4b62291a14c0a9e623c570532c88f84680ad6bb","source":{"kind":"arxiv","id":"2407.13853","version":3},"attestation_state":"computed","paper":{"title":"Forecasting GPU Performance for Deep Learning Training and Inference","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.PF"],"primary_cat":"cs.LG","authors_text":"Amar Phanishayee, Divya Mahajan, Seonho Lee","submitted_at":"2024-07-18T18:47:52Z","abstract_excerpt":"Deep learning kernels exhibit predictable memory accesses and compute patterns, making GPUs' parallel architecture well-suited for their execution. Software and runtime systems for GPUs are optimized to better utilize the stream multiprocessors, on-chip cache, and off-chip high-bandwidth memory. As deep learning models and GPUs evolve, access to newer GPUs is often limited, raising questions about the performance of new model architectures on existing GPUs, existing models on new GPUs, and new model architectures on new GPUs. To address these questions, we introduce NeuSight, a framework to pr"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2407.13853","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2024-07-18T18:47:52Z","cross_cats_sorted":["cs.PF"],"title_canon_sha256":"c781980cd79cb61ce433cd1496892a94fa8f29f0db132e57e95abb084b38d518","abstract_canon_sha256":"ee86f9d5bfdd43231409013c026e261a7c0d1c729b2a08a032e25e262776bcfe"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T09:47:56.027125Z","signature_b64":"c5t5v5JgYEEclLOioef6KDrJnkjiPxedvzSLRBH3/st9U3FrCu8pMkwcGIMgpxmdH9RVO6OIzaGTzf/gxln3Aw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"c7e2c0454d5766f902b61c5eb4b62291a14c0a9e623c570532c88f84680ad6bb","last_reissued_at":"2026-07-05T09:47:56.026656Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T09:47:56.026656Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Forecasting GPU Performance for Deep Learning Training and Inference","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.PF"],"primary_cat":"cs.LG","authors_text":"Amar Phanishayee, Divya Mahajan, Seonho Lee","submitted_at":"2024-07-18T18:47:52Z","abstract_excerpt":"Deep learning kernels exhibit predictable memory accesses and compute patterns, making GPUs' parallel architecture well-suited for their execution. Software and runtime systems for GPUs are optimized to better utilize the stream multiprocessors, on-chip cache, and off-chip high-bandwidth memory. As deep learning models and GPUs evolve, access to newer GPUs is often limited, raising questions about the performance of new model architectures on existing GPUs, existing models on new GPUs, and new model architectures on new GPUs. To address these questions, we introduce NeuSight, a framework to pr"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2407.13853","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2407.13853/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2407.13853","created_at":"2026-07-05T09:47:56.026715+00:00"},{"alias_kind":"arxiv_version","alias_value":"2407.13853v3","created_at":"2026-07-05T09:47:56.026715+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2407.13853","created_at":"2026-07-05T09:47:56.026715+00:00"},{"alias_kind":"pith_short_12","alias_value":"Y7RMARKNK5TP","created_at":"2026-07-05T09:47:56.026715+00:00"},{"alias_kind":"pith_short_16","alias_value":"Y7RMARKNK5TPSAVW","created_at":"2026-07-05T09:47:56.026715+00:00"},{"alias_kind":"pith_short_8","alias_value":"Y7RMARKN","created_at":"2026-07-05T09:47:56.026715+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2506.09275","citing_title":"A Survey of End-to-End Modeling for Distributed DNN Training: Workloads, Simulators, and TCO","ref_index":77,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/Y7RMARKNK5TPSAVWDRPLJNRCSG","json":"https://pith.science/pith/Y7RMARKNK5TPSAVWDRPLJNRCSG.json","graph_json":"https://pith.science/api/pith-number/Y7RMARKNK5TPSAVWDRPLJNRCSG/graph.json","events_json":"https://pith.science/api/pith-number/Y7RMARKNK5TPSAVWDRPLJNRCSG/events.json","paper":"https://pith.science/paper/Y7RMARKN"},"agent_actions":{"view_html":"https://pith.science/pith/Y7RMARKNK5TPSAVWDRPLJNRCSG","download_json":"https://pith.science/pith/Y7RMARKNK5TPSAVWDRPLJNRCSG.json","view_paper":"https://pith.science/paper/Y7RMARKN","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2407.13853&json=true","fetch_graph":"https://pith.science/api/pith-number/Y7RMARKNK5TPSAVWDRPLJNRCSG/graph.json","fetch_events":"https://pith.science/api/pith-number/Y7RMARKNK5TPSAVWDRPLJNRCSG/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/Y7RMARKNK5TPSAVWDRPLJNRCSG/action/timestamp_anchor","attest_storage":"https://pith.science/pith/Y7RMARKNK5TPSAVWDRPLJNRCSG/action/storage_attestation","attest_author":"https://pith.science/pith/Y7RMARKNK5TPSAVWDRPLJNRCSG/action/author_attestation","sign_citation":"https://pith.science/pith/Y7RMARKNK5TPSAVWDRPLJNRCSG/action/citation_signature","submit_replication":"https://pith.science/pith/Y7RMARKNK5TPSAVWDRPLJNRCSG/action/replication_record"}},"created_at":"2026-07-05T09:47:56.026715+00:00","updated_at":"2026-07-05T09:47:56.026715+00:00"}