{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:KYVQLUQOKCIO2LGRTHBHGYMWQX","short_pith_number":"pith:KYVQLUQO","schema_version":"1.0","canonical_sha256":"562b05d20e5090ed2cd199c273619685e4c9fac89fa07b911da4df8c6a1482ef","source":{"kind":"arxiv","id":"2301.13816","version":4},"attestation_state":"computed","paper":{"title":"Execution-based Code Generation using Deep Reinforcement Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.PL"],"primary_cat":"cs.LG","authors_text":"Aneesh Jain, Chandan K. Reddy, Parshin Shojaee, Sindhu Tipirneni","submitted_at":"2023-01-31T18:02:26Z","abstract_excerpt":"The utilization of programming language (PL) models, pre-trained on large-scale code corpora, as a means of automating software engineering processes has demonstrated considerable potential in streamlining various code generation tasks such as code completion, code translation, and program synthesis. However, current approaches mainly rely on supervised fine-tuning objectives borrowed from text generation, neglecting unique sequence-level characteristics of code, including but not limited to compilability as well as syntactic and functional correctness. To address this limitation, we propose P"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2301.13816","kind":"arxiv","version":4},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2023-01-31T18:02:26Z","cross_cats_sorted":["cs.AI","cs.CL","cs.PL"],"title_canon_sha256":"8f0f53d33017c1f472781591f603d1e002c0e4216d23d867cf596d1dd6847b86","abstract_canon_sha256":"238f38e48c0d4abbb6ec0d1878082da31938298695c9a5c1a9afbf434dcd37a0"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:32:53.077061Z","signature_b64":"Di8nKY6uWFi55DEt2kVFZu2c62g1pKAv1rUP05tJ9vsdm4oLhCsjvyA55CRX8Kn0LxTuVb6ks+yD26Db7EbgAw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"562b05d20e5090ed2cd199c273619685e4c9fac89fa07b911da4df8c6a1482ef","last_reissued_at":"2026-07-05T06:32:53.076501Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:32:53.076501Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Execution-based Code Generation using Deep Reinforcement Learning","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.PL"],"primary_cat":"cs.LG","authors_text":"Aneesh Jain, Chandan K. Reddy, Parshin Shojaee, Sindhu Tipirneni","submitted_at":"2023-01-31T18:02:26Z","abstract_excerpt":"The utilization of programming language (PL) models, pre-trained on large-scale code corpora, as a means of automating software engineering processes has demonstrated considerable potential in streamlining various code generation tasks such as code completion, code translation, and program synthesis. However, current approaches mainly rely on supervised fine-tuning objectives borrowed from text generation, neglecting unique sequence-level characteristics of code, including but not limited to compilability as well as syntactic and functional correctness. To address this limitation, we propose P"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2301.13816","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2301.13816/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2301.13816","created_at":"2026-07-05T06:32:53.076584+00:00"},{"alias_kind":"arxiv_version","alias_value":"2301.13816v4","created_at":"2026-07-05T06:32:53.076584+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2301.13816","created_at":"2026-07-05T06:32:53.076584+00:00"},{"alias_kind":"pith_short_12","alias_value":"KYVQLUQOKCIO","created_at":"2026-07-05T06:32:53.076584+00:00"},{"alias_kind":"pith_short_16","alias_value":"KYVQLUQOKCIO2LGR","created_at":"2026-07-05T06:32:53.076584+00:00"},{"alias_kind":"pith_short_8","alias_value":"KYVQLUQO","created_at":"2026-07-05T06:32:53.076584+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":14,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.21943","citing_title":"Modularized Reinforcement Learning on LLMs: From MDP Creation to Exploration and Learning","ref_index":176,"is_internal_anchor":false},{"citing_arxiv_id":"2607.01867","citing_title":"An Exploratory Study on LLM-Generated Code and Comments in Code Repositories","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2606.29999","citing_title":"AlgoSkill: Learning to Design Algorithms by Scheduling Human-Like Skills","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2605.25638","citing_title":"Reinforcement Learning from Denoising Feedback","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2606.03130","citing_title":"Synthetic Hallucinations, Real Gains: Hard Negatives from Frontier Models for FIM Hallucination Mitigation","ref_index":13,"is_internal_anchor":false},{"citing_arxiv_id":"2605.21467","citing_title":"DelTA: Discriminative Token Credit Assignment for Reinforcement Learning from Verifiable Rewards","ref_index":70,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17174","citing_title":"Beyond Execution: Static-Analysis Rewards and Hint-Conditioned Diffusion RL for Code Generation","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2406.00515","citing_title":"A Survey on Large Language Models for Code Generation","ref_index":243,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09134","citing_title":"BoostAPR: Boosting Automated Program Repair via Execution-Grounded Reinforcement Learning with Dual Reward Models","ref_index":34,"is_internal_anchor":false},{"citing_arxiv_id":"2605.09134","citing_title":"BoostAPR: Boosting Automated Program Repair via Execution-Grounded Reinforcement Learning with Dual Reward Models","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2605.00433","citing_title":"Improving LLM Code Generation via Requirement-Aware Curriculum Reinforcement Learning","ref_index":46,"is_internal_anchor":false},{"citing_arxiv_id":"2604.05560","citing_title":"An Iterative Test-and-Repair Framework for Competitive Code Generation","ref_index":48,"is_internal_anchor":false},{"citing_arxiv_id":"2604.18027","citing_title":"CodePivot: Bootstrapping Multilingual Transpilation in LLMs via Reinforcement Learning without Parallel Corpora","ref_index":59,"is_internal_anchor":false},{"citing_arxiv_id":"2604.17184","citing_title":"SynthFix: Adaptive Neuro-Symbolic Code Vulnerability Repair","ref_index":41,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/KYVQLUQOKCIO2LGRTHBHGYMWQX","json":"https://pith.science/pith/KYVQLUQOKCIO2LGRTHBHGYMWQX.json","graph_json":"https://pith.science/api/pith-number/KYVQLUQOKCIO2LGRTHBHGYMWQX/graph.json","events_json":"https://pith.science/api/pith-number/KYVQLUQOKCIO2LGRTHBHGYMWQX/events.json","paper":"https://pith.science/paper/KYVQLUQO"},"agent_actions":{"view_html":"https://pith.science/pith/KYVQLUQOKCIO2LGRTHBHGYMWQX","download_json":"https://pith.science/pith/KYVQLUQOKCIO2LGRTHBHGYMWQX.json","view_paper":"https://pith.science/paper/KYVQLUQO","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2301.13816&json=true","fetch_graph":"https://pith.science/api/pith-number/KYVQLUQOKCIO2LGRTHBHGYMWQX/graph.json","fetch_events":"https://pith.science/api/pith-number/KYVQLUQOKCIO2LGRTHBHGYMWQX/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/KYVQLUQOKCIO2LGRTHBHGYMWQX/action/timestamp_anchor","attest_storage":"https://pith.science/pith/KYVQLUQOKCIO2LGRTHBHGYMWQX/action/storage_attestation","attest_author":"https://pith.science/pith/KYVQLUQOKCIO2LGRTHBHGYMWQX/action/author_attestation","sign_citation":"https://pith.science/pith/KYVQLUQOKCIO2LGRTHBHGYMWQX/action/citation_signature","submit_replication":"https://pith.science/pith/KYVQLUQOKCIO2LGRTHBHGYMWQX/action/replication_record"}},"created_at":"2026-07-05T06:32:53.076584+00:00","updated_at":"2026-07-05T06:32:53.076584+00:00"}