{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:H4PN3PI2WKOFRNZWXYNGL7IOCL","short_pith_number":"pith:H4PN3PI2","schema_version":"1.0","canonical_sha256":"3f1eddbd1ab29c58b736be1a65fd0e12d0b352a4f2174b2fd4193b0fc450b03f","source":{"kind":"arxiv","id":"2307.04349","version":2},"attestation_state":"computed","paper":{"title":"RLTF: Reinforcement Learning from Unit Test Feedback","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL","cs.LG"],"primary_cat":"cs.AI","authors_text":"Deheng Ye, Jiate Liu, Kaiwen Xiao, Qiang Fu, Wei Yang, Xiao Han, Yiqin Zhu","submitted_at":"2023-07-10T05:18:18Z","abstract_excerpt":"The goal of program synthesis, or code generation, is to generate executable code based on given descriptions. Recently, there has been an increasing number of studies employing reinforcement learning (RL) to improve the performance of large language models (LLMs) for code. However, current representative works either rely solely on offline frameworks, limiting the exploration of new sample spaces, or fall short in the utilization of unit test signals, not accounting for specific error locations within the code. To address these issues, we propose RLTF, i.e., Reinforcement Learning from Unit T"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2307.04349","kind":"arxiv","version":2},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.AI","submitted_at":"2023-07-10T05:18:18Z","cross_cats_sorted":["cs.CL","cs.LG"],"title_canon_sha256":"9cad5cf687ca353362305f4bbe1a3609101087681415a9447702deabf30a26fa","abstract_canon_sha256":"a3022175f1f029fde4afb8460c1d737c9b0d12610ed66302c5de23d253ca203d"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:11:57.664192Z","signature_b64":"wc6jLBABSrFvNsuJKrdpRFnAmy97DXLobmo3goFk1j8vHwS2Wws8xQIav+HPD6ceZetJuBNq6YiJq970S17mCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"3f1eddbd1ab29c58b736be1a65fd0e12d0b352a4f2174b2fd4193b0fc450b03f","last_reissued_at":"2026-07-05T07:11:57.663692Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:11:57.663692Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"RLTF: Reinforcement Learning from Unit Test Feedback","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL","cs.LG"],"primary_cat":"cs.AI","authors_text":"Deheng Ye, Jiate Liu, Kaiwen Xiao, Qiang Fu, Wei Yang, Xiao Han, Yiqin Zhu","submitted_at":"2023-07-10T05:18:18Z","abstract_excerpt":"The goal of program synthesis, or code generation, is to generate executable code based on given descriptions. Recently, there has been an increasing number of studies employing reinforcement learning (RL) to improve the performance of large language models (LLMs) for code. However, current representative works either rely solely on offline frameworks, limiting the exploration of new sample spaces, or fall short in the utilization of unit test signals, not accounting for specific error locations within the code. To address these issues, we propose RLTF, i.e., Reinforcement Learning from Unit T"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2307.04349","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2307.04349/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2307.04349","created_at":"2026-07-05T07:11:57.663751+00:00"},{"alias_kind":"arxiv_version","alias_value":"2307.04349v2","created_at":"2026-07-05T07:11:57.663751+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2307.04349","created_at":"2026-07-05T07:11:57.663751+00:00"},{"alias_kind":"pith_short_12","alias_value":"H4PN3PI2WKOF","created_at":"2026-07-05T07:11:57.663751+00:00"},{"alias_kind":"pith_short_16","alias_value":"H4PN3PI2WKOFRNZW","created_at":"2026-07-05T07:11:57.663751+00:00"},{"alias_kind":"pith_short_8","alias_value":"H4PN3PI2","created_at":"2026-07-05T07:11:57.663751+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.28430","citing_title":"Building to the Test: Coding Agents Deliver What You Check, Not What You Requested","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2605.18747","citing_title":"Code as Agent Harness","ref_index":102,"is_internal_anchor":false},{"citing_arxiv_id":"2406.00515","citing_title":"A Survey on Large Language Models for Code Generation","ref_index":167,"is_internal_anchor":false},{"citing_arxiv_id":"2605.11680","citing_title":"ShapeCodeBench: A Renewable Benchmark for Perception-to-Program Reconstruction of Synthetic Shape Scenes","ref_index":11,"is_internal_anchor":false},{"citing_arxiv_id":"2605.02913","citing_title":"Generate, Filter, Control, Replay: A Comprehensive Survey of Rollout Strategies for LLM Reinforcement Learning","ref_index":76,"is_internal_anchor":false},{"citing_arxiv_id":"2604.05560","citing_title":"An Iterative Test-and-Repair Framework for Competitive Code Generation","ref_index":34,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/H4PN3PI2WKOFRNZWXYNGL7IOCL","json":"https://pith.science/pith/H4PN3PI2WKOFRNZWXYNGL7IOCL.json","graph_json":"https://pith.science/api/pith-number/H4PN3PI2WKOFRNZWXYNGL7IOCL/graph.json","events_json":"https://pith.science/api/pith-number/H4PN3PI2WKOFRNZWXYNGL7IOCL/events.json","paper":"https://pith.science/paper/H4PN3PI2"},"agent_actions":{"view_html":"https://pith.science/pith/H4PN3PI2WKOFRNZWXYNGL7IOCL","download_json":"https://pith.science/pith/H4PN3PI2WKOFRNZWXYNGL7IOCL.json","view_paper":"https://pith.science/paper/H4PN3PI2","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2307.04349&json=true","fetch_graph":"https://pith.science/api/pith-number/H4PN3PI2WKOFRNZWXYNGL7IOCL/graph.json","fetch_events":"https://pith.science/api/pith-number/H4PN3PI2WKOFRNZWXYNGL7IOCL/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/H4PN3PI2WKOFRNZWXYNGL7IOCL/action/timestamp_anchor","attest_storage":"https://pith.science/pith/H4PN3PI2WKOFRNZWXYNGL7IOCL/action/storage_attestation","attest_author":"https://pith.science/pith/H4PN3PI2WKOFRNZWXYNGL7IOCL/action/author_attestation","sign_citation":"https://pith.science/pith/H4PN3PI2WKOFRNZWXYNGL7IOCL/action/citation_signature","submit_replication":"https://pith.science/pith/H4PN3PI2WKOFRNZWXYNGL7IOCL/action/replication_record"}},"created_at":"2026-07-05T07:11:57.663751+00:00","updated_at":"2026-07-05T07:11:57.663751+00:00"}