{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:JLOBCHJIZMQQFO65SF2BZDDMGU","short_pith_number":"pith:JLOBCHJI","schema_version":"1.0","canonical_sha256":"4adc111d28cb2102bbdd91741c8c6c3534f8c127a4d23a3868c0a8d4fb82cf03","source":{"kind":"arxiv","id":"2501.08071","version":1},"attestation_state":"computed","paper":{"title":"CuAsmRL: Optimizing GPU SASS Schedules via Deep Reinforcement Learning","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.AR","authors_text":"Eiko Yoneki, Guoliang He","submitted_at":"2025-01-14T12:36:18Z","abstract_excerpt":"Large language models (LLMs) are remarked by their substantial computational requirements. To mitigate the cost, researchers develop specialized CUDA kernels, which often fuse several tensor operations to maximize the utilization of GPUs as much as possible. However, those specialized kernels may still leave performance on the table as CUDA assembly experts show that manual optimization of GPU SASS schedules can lead to better performance, and trial-and-error is largely employed to manually find the best GPU SASS schedules.\n  In this work, we employ an automatic approach to optimize GPU SASS s"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2501.08071","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by-sa/4.0/","primary_cat":"cs.AR","submitted_at":"2025-01-14T12:36:18Z","cross_cats_sorted":["cs.LG"],"title_canon_sha256":"1a4e38a0b356beedcefb41d179f87e70441d64eb6661a79268bc7755ed3b2f1d","abstract_canon_sha256":"32aa8763b025c2b453f4358350e944a6f60ed70a991d526954a532f4be43e6f5"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:00:55.298716Z","signature_b64":"Ulk/DNorI2vZ4b0693FtE5JG2QYp7ixBBAzmu17RQwfPOcEsPza47pUnzlO3oWdYbLXzz8ORaixOxgrucdWXBg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"4adc111d28cb2102bbdd91741c8c6c3534f8c127a4d23a3868c0a8d4fb82cf03","last_reissued_at":"2026-07-05T10:00:55.298355Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:00:55.298355Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"CuAsmRL: Optimizing GPU SASS Schedules via Deep Reinforcement Learning","license":"http://creativecommons.org/licenses/by-sa/4.0/","headline":"","cross_cats":["cs.LG"],"primary_cat":"cs.AR","authors_text":"Eiko Yoneki, Guoliang He","submitted_at":"2025-01-14T12:36:18Z","abstract_excerpt":"Large language models (LLMs) are remarked by their substantial computational requirements. To mitigate the cost, researchers develop specialized CUDA kernels, which often fuse several tensor operations to maximize the utilization of GPUs as much as possible. However, those specialized kernels may still leave performance on the table as CUDA assembly experts show that manual optimization of GPU SASS schedules can lead to better performance, and trial-and-error is largely employed to manually find the best GPU SASS schedules.\n  In this work, we employ an automatic approach to optimize GPU SASS s"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2501.08071","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2501.08071/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2501.08071","created_at":"2026-07-05T10:00:55.298412+00:00"},{"alias_kind":"arxiv_version","alias_value":"2501.08071v1","created_at":"2026-07-05T10:00:55.298412+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2501.08071","created_at":"2026-07-05T10:00:55.298412+00:00"},{"alias_kind":"pith_short_12","alias_value":"JLOBCHJIZMQQ","created_at":"2026-07-05T10:00:55.298412+00:00"},{"alias_kind":"pith_short_16","alias_value":"JLOBCHJIZMQQFO65","created_at":"2026-07-05T10:00:55.298412+00:00"},{"alias_kind":"pith_short_8","alias_value":"JLOBCHJI","created_at":"2026-07-05T10:00:55.298412+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/JLOBCHJIZMQQFO65SF2BZDDMGU","json":"https://pith.science/pith/JLOBCHJIZMQQFO65SF2BZDDMGU.json","graph_json":"https://pith.science/api/pith-number/JLOBCHJIZMQQFO65SF2BZDDMGU/graph.json","events_json":"https://pith.science/api/pith-number/JLOBCHJIZMQQFO65SF2BZDDMGU/events.json","paper":"https://pith.science/paper/JLOBCHJI"},"agent_actions":{"view_html":"https://pith.science/pith/JLOBCHJIZMQQFO65SF2BZDDMGU","download_json":"https://pith.science/pith/JLOBCHJIZMQQFO65SF2BZDDMGU.json","view_paper":"https://pith.science/paper/JLOBCHJI","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2501.08071&json=true","fetch_graph":"https://pith.science/api/pith-number/JLOBCHJIZMQQFO65SF2BZDDMGU/graph.json","fetch_events":"https://pith.science/api/pith-number/JLOBCHJIZMQQFO65SF2BZDDMGU/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/JLOBCHJIZMQQFO65SF2BZDDMGU/action/timestamp_anchor","attest_storage":"https://pith.science/pith/JLOBCHJIZMQQFO65SF2BZDDMGU/action/storage_attestation","attest_author":"https://pith.science/pith/JLOBCHJIZMQQFO65SF2BZDDMGU/action/author_attestation","sign_citation":"https://pith.science/pith/JLOBCHJIZMQQFO65SF2BZDDMGU/action/citation_signature","submit_replication":"https://pith.science/pith/JLOBCHJIZMQQFO65SF2BZDDMGU/action/replication_record"}},"created_at":"2026-07-05T10:00:55.298412+00:00","updated_at":"2026-07-05T10:00:55.298412+00:00"}