{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2026:MXK4DYRZDKBDQP3FOIMXTKSTJB","short_pith_number":"pith:MXK4DYRZ","schema_version":"1.0","canonical_sha256":"65d5c1e2391a82383f65721979aa534848d9453c3e7b493e7bac6da37cd5812e","source":{"kind":"arxiv","id":"2607.25970","version":1},"attestation_state":"computed","paper":{"title":"Reinforcement Learning for Code Optimization","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Benoit Sagot, Gabriel Synnaeve, Juliette Decugis, Kunhao Zheng, Pierre Chambon","submitted_at":"2026-07-28T16:52:31Z","abstract_excerpt":"RL for code correctness is now established: have the model generate a program, run it against hidden test cases, and reward solutions that pass. Extending this to code optimization seems straightforward: just add execution time to the reward. But in practice, once timing drives the reward, small problems in measurement noise, reward sparsity, or GRPO instability overwhelm the signal and make RL fail: generated solutions are barely faster, and more of them can fail. We make execution time learnable through three stages: (1) how code is tested, by building DMC-Optim with large optimization tests"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2607.25970","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2026-07-28T16:52:31Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"08e3ace4917d77ab72320b415f471f7bddd659deb9d4b3f209bff3fd6466bd5a","abstract_canon_sha256":"766f7f31053404fc7cd62c958c5712e8b521c08cd35b6956cb22da3b033a9d8b"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-29T01:26:17.105218Z","signature_b64":"zEjmMhjw6/BQgErqMmgIBpZXRrFMSCtbJwFuC+DqHa/T816i4s/uPsBbzkoEAiLj50GNN9NwxBvCLXkNa15eCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"65d5c1e2391a82383f65721979aa534848d9453c3e7b493e7bac6da37cd5812e","last_reissued_at":"2026-07-29T01:26:17.104350Z","signature_status":"signed_v1","first_computed_at":"2026-07-29T01:26:17.104350Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Reinforcement Learning for Code Optimization","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.LG","authors_text":"Benoit Sagot, Gabriel Synnaeve, Juliette Decugis, Kunhao Zheng, Pierre Chambon","submitted_at":"2026-07-28T16:52:31Z","abstract_excerpt":"RL for code correctness is now established: have the model generate a program, run it against hidden test cases, and reward solutions that pass. Extending this to code optimization seems straightforward: just add execution time to the reward. But in practice, once timing drives the reward, small problems in measurement noise, reward sparsity, or GRPO instability overwhelm the signal and make RL fail: generated solutions are barely faster, and more of them can fail. We make execution time learnable through three stages: (1) how code is tested, by building DMC-Optim with large optimization tests"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2607.25970","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2607.25970/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2607.25970","created_at":"2026-07-29T01:26:17.104785+00:00"},{"alias_kind":"arxiv_version","alias_value":"2607.25970v1","created_at":"2026-07-29T01:26:17.104785+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2607.25970","created_at":"2026-07-29T01:26:17.104785+00:00"},{"alias_kind":"pith_short_12","alias_value":"MXK4DYRZDKBD","created_at":"2026-07-29T01:26:17.104785+00:00"},{"alias_kind":"pith_short_16","alias_value":"MXK4DYRZDKBDQP3F","created_at":"2026-07-29T01:26:17.104785+00:00"},{"alias_kind":"pith_short_8","alias_value":"MXK4DYRZ","created_at":"2026-07-29T01:26:17.104785+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":0,"internal_anchor_count":0,"sample":[]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/MXK4DYRZDKBDQP3FOIMXTKSTJB","json":"https://pith.science/pith/MXK4DYRZDKBDQP3FOIMXTKSTJB.json","graph_json":"https://pith.science/api/pith-number/MXK4DYRZDKBDQP3FOIMXTKSTJB/graph.json","events_json":"https://pith.science/api/pith-number/MXK4DYRZDKBDQP3FOIMXTKSTJB/events.json","paper":"https://pith.science/paper/MXK4DYRZ"},"agent_actions":{"view_html":"https://pith.science/pith/MXK4DYRZDKBDQP3FOIMXTKSTJB","download_json":"https://pith.science/pith/MXK4DYRZDKBDQP3FOIMXTKSTJB.json","view_paper":"https://pith.science/paper/MXK4DYRZ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2607.25970&json=true","fetch_graph":"https://pith.science/api/pith-number/MXK4DYRZDKBDQP3FOIMXTKSTJB/graph.json","fetch_events":"https://pith.science/api/pith-number/MXK4DYRZDKBDQP3FOIMXTKSTJB/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/MXK4DYRZDKBDQP3FOIMXTKSTJB/action/timestamp_anchor","attest_storage":"https://pith.science/pith/MXK4DYRZDKBDQP3FOIMXTKSTJB/action/storage_attestation","attest_author":"https://pith.science/pith/MXK4DYRZDKBDQP3FOIMXTKSTJB/action/author_attestation","sign_citation":"https://pith.science/pith/MXK4DYRZDKBDQP3FOIMXTKSTJB/action/citation_signature","submit_replication":"https://pith.science/pith/MXK4DYRZDKBDQP3FOIMXTKSTJB/action/replication_record"}},"created_at":"2026-07-29T01:26:17.104785+00:00","updated_at":"2026-07-29T01:26:17.104785+00:00"}