{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:J2QANFYJU2LPQJMVCDNWMIFJGQ","short_pith_number":"pith:J2QANFYJ","schema_version":"1.0","canonical_sha256":"4ea0069709a696f8259510db6620a93405f261d8ca5c841c6cefcd7e49eed988","source":{"kind":"arxiv","id":"2302.08468","version":3},"attestation_state":"computed","paper":{"title":"LEVER: Learning to Verify Language-to-Code Generation with Execution","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.CL","cs.PL","cs.SE"],"primary_cat":"cs.LG","authors_text":"Ansong Ni, Dragomir Radev, Sida I. Wang, Srini Iyer, Ves Stoyanov, Wen-tau Yih, Xi Victoria Lin","submitted_at":"2023-02-16T18:23:22Z","abstract_excerpt":"The advent of large language models trained on code (code LLMs) has led to significant progress in language-to-code generation. State-of-the-art approaches in this area combine LLM decoding with sample pruning and reranking using test cases or heuristics based on the execution results. However, it is challenging to obtain test cases for many real-world language-to-code applications, and heuristics cannot well capture the semantic features of the execution results, such as data type and value range, which often indicates the correctness of the program. In this work, we propose LEVER, a simple a"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2302.08468","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","primary_cat":"cs.LG","submitted_at":"2023-02-16T18:23:22Z","cross_cats_sorted":["cs.CL","cs.PL","cs.SE"],"title_canon_sha256":"5fc6e8ecf937fb9effb1c496de1821a4e9252c3fecc90dd312d5a387c0ba8b88","abstract_canon_sha256":"9db7f5fa52e4368942fc6c37dfbc81f4949abcdac0d7b0127523f415b40f7d34"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:46:46.083388Z","signature_b64":"wg+vlx52uUFH4fdPA+M7aUZxshzmk1fZpRtwGWhZk/gxJjGCr7yccdIAdTIpigbAOt5C1nabiCmfW7K/z/CODw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"4ea0069709a696f8259510db6620a93405f261d8ca5c841c6cefcd7e49eed988","last_reissued_at":"2026-07-05T06:46:46.082926Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:46:46.082926Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"LEVER: Learning to Verify Language-to-Code Generation with Execution","license":"http://creativecommons.org/licenses/by-nc-sa/4.0/","headline":"","cross_cats":["cs.CL","cs.PL","cs.SE"],"primary_cat":"cs.LG","authors_text":"Ansong Ni, Dragomir Radev, Sida I. Wang, Srini Iyer, Ves Stoyanov, Wen-tau Yih, Xi Victoria Lin","submitted_at":"2023-02-16T18:23:22Z","abstract_excerpt":"The advent of large language models trained on code (code LLMs) has led to significant progress in language-to-code generation. State-of-the-art approaches in this area combine LLM decoding with sample pruning and reranking using test cases or heuristics based on the execution results. However, it is challenging to obtain test cases for many real-world language-to-code applications, and heuristics cannot well capture the semantic features of the execution results, such as data type and value range, which often indicates the correctness of the program. In this work, we propose LEVER, a simple a"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2302.08468","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2302.08468/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2302.08468","created_at":"2026-07-05T06:46:46.082987+00:00"},{"alias_kind":"arxiv_version","alias_value":"2302.08468v3","created_at":"2026-07-05T06:46:46.082987+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2302.08468","created_at":"2026-07-05T06:46:46.082987+00:00"},{"alias_kind":"pith_short_12","alias_value":"J2QANFYJU2LP","created_at":"2026-07-05T06:46:46.082987+00:00"},{"alias_kind":"pith_short_16","alias_value":"J2QANFYJU2LPQJMV","created_at":"2026-07-05T06:46:46.082987+00:00"},{"alias_kind":"pith_short_8","alias_value":"J2QANFYJ","created_at":"2026-07-05T06:46:46.082987+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.16829","citing_title":"Constrained Code Generation with Discrete Diffusion","ref_index":12,"is_internal_anchor":false},{"citing_arxiv_id":"2511.01008","citing_title":"MARS-SQL: A multi-agent reinforcement learning framework for Text-to-SQL","ref_index":22,"is_internal_anchor":false},{"citing_arxiv_id":"2304.05128","citing_title":"Teaching Large Language Models to Self-Debug","ref_index":112,"is_internal_anchor":false},{"citing_arxiv_id":"2605.07935","citing_title":"TraceFix: Repairing Agent Coordination Protocols with TLA+ Counterexamples","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2305.16291","citing_title":"Voyager: An Open-Ended Embodied Agent with Large Language Models","ref_index":91,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/J2QANFYJU2LPQJMVCDNWMIFJGQ","json":"https://pith.science/pith/J2QANFYJU2LPQJMVCDNWMIFJGQ.json","graph_json":"https://pith.science/api/pith-number/J2QANFYJU2LPQJMVCDNWMIFJGQ/graph.json","events_json":"https://pith.science/api/pith-number/J2QANFYJU2LPQJMVCDNWMIFJGQ/events.json","paper":"https://pith.science/paper/J2QANFYJ"},"agent_actions":{"view_html":"https://pith.science/pith/J2QANFYJU2LPQJMVCDNWMIFJGQ","download_json":"https://pith.science/pith/J2QANFYJU2LPQJMVCDNWMIFJGQ.json","view_paper":"https://pith.science/paper/J2QANFYJ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2302.08468&json=true","fetch_graph":"https://pith.science/api/pith-number/J2QANFYJU2LPQJMVCDNWMIFJGQ/graph.json","fetch_events":"https://pith.science/api/pith-number/J2QANFYJU2LPQJMVCDNWMIFJGQ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/J2QANFYJU2LPQJMVCDNWMIFJGQ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/J2QANFYJU2LPQJMVCDNWMIFJGQ/action/storage_attestation","attest_author":"https://pith.science/pith/J2QANFYJU2LPQJMVCDNWMIFJGQ/action/author_attestation","sign_citation":"https://pith.science/pith/J2QANFYJU2LPQJMVCDNWMIFJGQ/action/citation_signature","submit_replication":"https://pith.science/pith/J2QANFYJU2LPQJMVCDNWMIFJGQ/action/replication_record"}},"created_at":"2026-07-05T06:46:46.082987+00:00","updated_at":"2026-07-05T06:46:46.082987+00:00"}