{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:HT2PVKIT2OJINFIQ4HP4I6BZF5","short_pith_number":"pith:HT2PVKIT","schema_version":"1.0","canonical_sha256":"3cf4faa913d392869510e1dfc478392f644ad8befa1914ad08c1d61bfe102d4f","source":{"kind":"arxiv","id":"2507.14111","version":11},"attestation_state":"computed","paper":{"title":"CUDA-L1: Improving CUDA Optimization via Contrastive Reinforcement Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.DC","cs.LG"],"primary_cat":"cs.AI","authors_text":"Albert Wang, Chris Shum, Guoyin Wang, Jiwei Li, Xiaoya Li","submitted_at":"2025-07-18T17:43:56Z","abstract_excerpt":"The exponential growth in demand for GPU computing resources has created an urgent need for automated CUDA optimization strategies. While recent advances in LLMs show promise for code generation, current SOTA models achieve low success rates in improving CUDA speed. In this paper, we introduce CUDA-L1, an automated reinforcement learning framework for CUDA optimization that employs a novel contrastive RL algorithm.\n  CUDA-L1 achieves significant performance improvements on the CUDA optimization task: trained on A100, it delivers an average speedup of x3.12 with a median speedup of x1.42 agains"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2507.14111","kind":"arxiv","version":11},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.AI","submitted_at":"2025-07-18T17:43:56Z","cross_cats_sorted":["cs.DC","cs.LG"],"title_canon_sha256":"b844956ca4d277693216c3d91bda03b6c991a3be6804feb22bdf5af4bf13dba8","abstract_canon_sha256":"6b8ecdbfda9ab55c156747b41eb1e86a64d90d6aefa48211fd02679d50e0484e"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-14T01:20:44.983641Z","signature_b64":"jgeudof8BTGWz7LG+Fu6mJDnzwVdjMNDMkTym+23vytxEis3G1+hKQdmtR23xfwIyJ8jt9s2+kr+OGr61VU7Dw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"3cf4faa913d392869510e1dfc478392f644ad8befa1914ad08c1d61bfe102d4f","last_reissued_at":"2026-07-14T01:20:44.982693Z","signature_status":"signed_v1","first_computed_at":"2026-07-14T01:20:44.982693Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"CUDA-L1: Improving CUDA Optimization via Contrastive Reinforcement Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.DC","cs.LG"],"primary_cat":"cs.AI","authors_text":"Albert Wang, Chris Shum, Guoyin Wang, Jiwei Li, Xiaoya Li","submitted_at":"2025-07-18T17:43:56Z","abstract_excerpt":"The exponential growth in demand for GPU computing resources has created an urgent need for automated CUDA optimization strategies. While recent advances in LLMs show promise for code generation, current SOTA models achieve low success rates in improving CUDA speed. In this paper, we introduce CUDA-L1, an automated reinforcement learning framework for CUDA optimization that employs a novel contrastive RL algorithm.\n  CUDA-L1 achieves significant performance improvements on the CUDA optimization task: trained on A100, it delivers an average speedup of x3.12 with a median speedup of x1.42 agains"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2507.14111","kind":"arxiv","version":11},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2507.14111/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2507.14111","created_at":"2026-07-14T01:20:44.983159+00:00"},{"alias_kind":"arxiv_version","alias_value":"2507.14111v11","created_at":"2026-07-14T01:20:44.983159+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2507.14111","created_at":"2026-07-14T01:20:44.983159+00:00"},{"alias_kind":"pith_short_12","alias_value":"HT2PVKIT2OJI","created_at":"2026-07-14T01:20:44.983159+00:00"},{"alias_kind":"pith_short_16","alias_value":"HT2PVKIT2OJINFIQ","created_at":"2026-07-14T01:20:44.983159+00:00"},{"alias_kind":"pith_short_8","alias_value":"HT2PVKIT","created_at":"2026-07-14T01:20:44.983159+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":11,"internal_anchor_count":11,"sample":[{"citing_arxiv_id":"2606.17518","citing_title":"SpecGen: Accelerating Agentic Kernel Optimization with Speculative Generation","ref_index":22,"is_internal_anchor":true},{"citing_arxiv_id":"2607.01590","citing_title":"Hawk: Harnessing Hardware-Aware Knowledge for High-Performance NPU Kernel Generation","ref_index":27,"is_internal_anchor":true},{"citing_arxiv_id":"2605.30359","citing_title":"Kernel Foundry: A Diagnosis-driven Evolutionary Kernel Optimizer with Multi-Experts","ref_index":10,"is_internal_anchor":true},{"citing_arxiv_id":"2605.25954","citing_title":"Step-TP: A Grounded, Step-Level Dataset with Chain-of-Thought Reasoning for LLM-Guided Tensor Program Optimization","ref_index":26,"is_internal_anchor":true},{"citing_arxiv_id":"2605.28213","citing_title":"Learning When to Optimize: Verified Optimization Skills from Expert GPU-Kernel Lineages","ref_index":16,"is_internal_anchor":true},{"citing_arxiv_id":"2605.29357","citing_title":"PassNet: Scaling Large Language Models for Graph Compiler Pass Generation","ref_index":12,"is_internal_anchor":true},{"citing_arxiv_id":"2605.23215","citing_title":"FastKernels: Benchmarking GPU Kernel Generation in Production","ref_index":11,"is_internal_anchor":true},{"citing_arxiv_id":"2603.23566","citing_title":"AscendOptimizer: Episodic Agent for Ascend NPU Operator Optimization","ref_index":17,"is_internal_anchor":true},{"citing_arxiv_id":"2603.28342","citing_title":"Kernel-Smith: A Unified Recipe for Evolutionary Kernel Optimization","ref_index":12,"is_internal_anchor":true},{"citing_arxiv_id":"2604.02721","citing_title":"GrandCode: Achieving Grandmaster Level in Competitive Programming via Agentic Reinforcement Learning","ref_index":15,"is_internal_anchor":true},{"citing_arxiv_id":"2604.16625","citing_title":"AdaExplore: Failure-Driven Adaptation and Diversity-Preserving Search for Efficient Kernel Generation","ref_index":9,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/HT2PVKIT2OJINFIQ4HP4I6BZF5","json":"https://pith.science/pith/HT2PVKIT2OJINFIQ4HP4I6BZF5.json","graph_json":"https://pith.science/api/pith-number/HT2PVKIT2OJINFIQ4HP4I6BZF5/graph.json","events_json":"https://pith.science/api/pith-number/HT2PVKIT2OJINFIQ4HP4I6BZF5/events.json","paper":"https://pith.science/paper/HT2PVKIT"},"agent_actions":{"view_html":"https://pith.science/pith/HT2PVKIT2OJINFIQ4HP4I6BZF5","download_json":"https://pith.science/pith/HT2PVKIT2OJINFIQ4HP4I6BZF5.json","view_paper":"https://pith.science/paper/HT2PVKIT","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2507.14111&json=true","fetch_graph":"https://pith.science/api/pith-number/HT2PVKIT2OJINFIQ4HP4I6BZF5/graph.json","fetch_events":"https://pith.science/api/pith-number/HT2PVKIT2OJINFIQ4HP4I6BZF5/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/HT2PVKIT2OJINFIQ4HP4I6BZF5/action/timestamp_anchor","attest_storage":"https://pith.science/pith/HT2PVKIT2OJINFIQ4HP4I6BZF5/action/storage_attestation","attest_author":"https://pith.science/pith/HT2PVKIT2OJINFIQ4HP4I6BZF5/action/author_attestation","sign_citation":"https://pith.science/pith/HT2PVKIT2OJINFIQ4HP4I6BZF5/action/citation_signature","submit_replication":"https://pith.science/pith/HT2PVKIT2OJINFIQ4HP4I6BZF5/action/replication_record"}},"created_at":"2026-07-14T01:20:44.983159+00:00","updated_at":"2026-07-14T01:20:44.983159+00:00"}