{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2025:PL2N6E2YH76QJCBP3NLYCXJG6U","short_pith_number":"pith:PL2N6E2Y","schema_version":"1.0","canonical_sha256":"7af4df13583ffd04882fdb57815d26f50574438aeec6bdc655459fdcf423b461","source":{"kind":"arxiv","id":"2507.06920","version":2},"attestation_state":"computed","paper":{"title":"Rethinking Verification for LLM Code Generation: From Generation to Testing","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Junnan Liu, Kai Chen, Maosong Cao, Minnan Luo, Songyang Zhang, Taolin Zhang, Wenwei Zhang, Zihan Ma","submitted_at":"2025-07-09T14:58:47Z","abstract_excerpt":"Large language models (LLMs) have recently achieved notable success in code-generation benchmarks such as HumanEval and LiveCodeBench. However, a detailed examination reveals that these evaluation suites often comprise only a limited number of homogeneous test cases, resulting in subtle faults going undetected. This not only artificially inflates measured performance but also compromises accurate reward estimation in reinforcement learning frameworks utilizing verifiable rewards (RLVR). To address these critical shortcomings, we systematically investigate the test-case generation (TCG) task by"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2507.06920","kind":"arxiv","version":2},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2025-07-09T14:58:47Z","cross_cats_sorted":[],"title_canon_sha256":"6755d82b10b92a8d76bb1b1686787e7340e5221e26071ed602665a3d363416df","abstract_canon_sha256":"5d3023ce98dcff42e1599d06d678f213fc36c72641cb83af41ded7c4119f943d"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T11:34:54.352712Z","signature_b64":"gLUnVMyqf81wBhuMcWw3SCTQOS1XptF1eJIE/fo0UhgDHHFMdFRWAFWDO6Im4pzrAhGj8t0Z2OZ5otD87aySDg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"7af4df13583ffd04882fdb57815d26f50574438aeec6bdc655459fdcf423b461","last_reissued_at":"2026-07-05T11:34:54.352233Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T11:34:54.352233Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Rethinking Verification for LLM Code Generation: From Generation to Testing","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.CL","authors_text":"Junnan Liu, Kai Chen, Maosong Cao, Minnan Luo, Songyang Zhang, Taolin Zhang, Wenwei Zhang, Zihan Ma","submitted_at":"2025-07-09T14:58:47Z","abstract_excerpt":"Large language models (LLMs) have recently achieved notable success in code-generation benchmarks such as HumanEval and LiveCodeBench. However, a detailed examination reveals that these evaluation suites often comprise only a limited number of homogeneous test cases, resulting in subtle faults going undetected. This not only artificially inflates measured performance but also compromises accurate reward estimation in reinforcement learning frameworks utilizing verifiable rewards (RLVR). To address these critical shortcomings, we systematically investigate the test-case generation (TCG) task by"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2507.06920","kind":"arxiv","version":2},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2507.06920/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2507.06920","created_at":"2026-07-05T11:34:54.352289+00:00"},{"alias_kind":"arxiv_version","alias_value":"2507.06920v2","created_at":"2026-07-05T11:34:54.352289+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2507.06920","created_at":"2026-07-05T11:34:54.352289+00:00"},{"alias_kind":"pith_short_12","alias_value":"PL2N6E2YH76Q","created_at":"2026-07-05T11:34:54.352289+00:00"},{"alias_kind":"pith_short_16","alias_value":"PL2N6E2YH76QJCBP","created_at":"2026-07-05T11:34:54.352289+00:00"},{"alias_kind":"pith_short_8","alias_value":"PL2N6E2Y","created_at":"2026-07-05T11:34:54.352289+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":1,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2608.01000","citing_title":"Judging Is Not Enumerating: Silent Omissions in LLM-Authored Acceptable Sets","ref_index":24,"is_internal_anchor":true}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/PL2N6E2YH76QJCBP3NLYCXJG6U","json":"https://pith.science/pith/PL2N6E2YH76QJCBP3NLYCXJG6U.json","graph_json":"https://pith.science/api/pith-number/PL2N6E2YH76QJCBP3NLYCXJG6U/graph.json","events_json":"https://pith.science/api/pith-number/PL2N6E2YH76QJCBP3NLYCXJG6U/events.json","paper":"https://pith.science/paper/PL2N6E2Y"},"agent_actions":{"view_html":"https://pith.science/pith/PL2N6E2YH76QJCBP3NLYCXJG6U","download_json":"https://pith.science/pith/PL2N6E2YH76QJCBP3NLYCXJG6U.json","view_paper":"https://pith.science/paper/PL2N6E2Y","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2507.06920&json=true","fetch_graph":"https://pith.science/api/pith-number/PL2N6E2YH76QJCBP3NLYCXJG6U/graph.json","fetch_events":"https://pith.science/api/pith-number/PL2N6E2YH76QJCBP3NLYCXJG6U/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/PL2N6E2YH76QJCBP3NLYCXJG6U/action/timestamp_anchor","attest_storage":"https://pith.science/pith/PL2N6E2YH76QJCBP3NLYCXJG6U/action/storage_attestation","attest_author":"https://pith.science/pith/PL2N6E2YH76QJCBP3NLYCXJG6U/action/author_attestation","sign_citation":"https://pith.science/pith/PL2N6E2YH76QJCBP3NLYCXJG6U/action/citation_signature","submit_replication":"https://pith.science/pith/PL2N6E2YH76QJCBP3NLYCXJG6U/action/replication_record"}},"created_at":"2026-07-05T11:34:54.352289+00:00","updated_at":"2026-07-05T11:34:54.352289+00:00"}