{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:5LL2BD22YNRWGYQWNP4DLIHRQI","short_pith_number":"pith:5LL2BD22","schema_version":"1.0","canonical_sha256":"ead7a08f5ac3636362166bf835a0f1820c3fdee894f952619a2db662deaa1926","source":{"kind":"arxiv","id":"2212.10264","version":1},"attestation_state":"computed","paper":{"title":"ReCode: Robustness Evaluation of Code Generation Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL","cs.SE"],"primary_cat":"cs.LG","authors_text":"Baishakhi Ray, Bing Xiang, Chenghao Yang, Dan Roth, Haifeng Qian, Mingyue Shang, Murali Krishna Ramanathan, Parminder Bhatia, Ramesh Nallapati, Samson Tan, Shiqi Wang, Varun Kumar, Zheng Li, Zijian Wang","submitted_at":"2022-12-20T14:11:31Z","abstract_excerpt":"Code generation models have achieved impressive performance. However, they tend to be brittle as slight edits to a prompt could lead to very different generations; these robustness properties, critical for user experience when deployed in real-life applications, are not well understood. Most existing works on robustness in text or code tasks have focused on classification, while robustness in generation tasks is an uncharted area and to date there is no comprehensive benchmark for robustness in code generation. In this paper, we propose ReCode, a comprehensive robustness evaluation benchmark f"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2212.10264","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.LG","submitted_at":"2022-12-20T14:11:31Z","cross_cats_sorted":["cs.CL","cs.SE"],"title_canon_sha256":"9b3995fbe855a9deb72bf9380b82cca510cfad05ab2dbb5338a1224fccc9f1d1","abstract_canon_sha256":"64d78d48f3d574930d6f6fe20c7733756ddd8600b49cda8398a79fa8eb195f1c"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T05:27:05.780394Z","signature_b64":"Qfj5LYNa5zKzd2VicoMuR3ArXdkk3yoEcydRCL+BbwWzb5jFfCAB1iZLMZRtZMJUYWojlz0DxacHlRa3AYS5BQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"ead7a08f5ac3636362166bf835a0f1820c3fdee894f952619a2db662deaa1926","last_reissued_at":"2026-07-05T05:27:05.779949Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T05:27:05.779949Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"ReCode: Robustness Evaluation of Code Generation Models","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.CL","cs.SE"],"primary_cat":"cs.LG","authors_text":"Baishakhi Ray, Bing Xiang, Chenghao Yang, Dan Roth, Haifeng Qian, Mingyue Shang, Murali Krishna Ramanathan, Parminder Bhatia, Ramesh Nallapati, Samson Tan, Shiqi Wang, Varun Kumar, Zheng Li, Zijian Wang","submitted_at":"2022-12-20T14:11:31Z","abstract_excerpt":"Code generation models have achieved impressive performance. However, they tend to be brittle as slight edits to a prompt could lead to very different generations; these robustness properties, critical for user experience when deployed in real-life applications, are not well understood. Most existing works on robustness in text or code tasks have focused on classification, while robustness in generation tasks is an uncharted area and to date there is no comprehensive benchmark for robustness in code generation. In this paper, we propose ReCode, a comprehensive robustness evaluation benchmark f"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2212.10264","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2212.10264/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2212.10264","created_at":"2026-07-05T05:27:05.780002+00:00"},{"alias_kind":"arxiv_version","alias_value":"2212.10264v1","created_at":"2026-07-05T05:27:05.780002+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2212.10264","created_at":"2026-07-05T05:27:05.780002+00:00"},{"alias_kind":"pith_short_12","alias_value":"5LL2BD22YNRW","created_at":"2026-07-05T05:27:05.780002+00:00"},{"alias_kind":"pith_short_16","alias_value":"5LL2BD22YNRWGYQW","created_at":"2026-07-05T05:27:05.780002+00:00"},{"alias_kind":"pith_short_8","alias_value":"5LL2BD22","created_at":"2026-07-05T05:27:05.780002+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":5,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2605.14415","citing_title":"SWE-Chain: Benchmarking Coding Agents on Chained Release-Level Package Upgrades","ref_index":37,"is_internal_anchor":false},{"citing_arxiv_id":"2406.00515","citing_title":"A Survey on Large Language Models for Code Generation","ref_index":268,"is_internal_anchor":false},{"citing_arxiv_id":"2605.05267","citing_title":"Bridging Generation and Training: A Systematic Review of Quality Issues in LLMs for Code","ref_index":127,"is_internal_anchor":false},{"citing_arxiv_id":"2604.12214","citing_title":"Structural Anchors and Reasoning Fragility:Understanding CoT Robustness in LLM4Code","ref_index":5,"is_internal_anchor":false},{"citing_arxiv_id":"2403.07974","citing_title":"LiveCodeBench: Holistic and Contamination Free Evaluation of Large Language Models for Code","ref_index":13,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/5LL2BD22YNRWGYQWNP4DLIHRQI","json":"https://pith.science/pith/5LL2BD22YNRWGYQWNP4DLIHRQI.json","graph_json":"https://pith.science/api/pith-number/5LL2BD22YNRWGYQWNP4DLIHRQI/graph.json","events_json":"https://pith.science/api/pith-number/5LL2BD22YNRWGYQWNP4DLIHRQI/events.json","paper":"https://pith.science/paper/5LL2BD22"},"agent_actions":{"view_html":"https://pith.science/pith/5LL2BD22YNRWGYQWNP4DLIHRQI","download_json":"https://pith.science/pith/5LL2BD22YNRWGYQWNP4DLIHRQI.json","view_paper":"https://pith.science/paper/5LL2BD22","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2212.10264&json=true","fetch_graph":"https://pith.science/api/pith-number/5LL2BD22YNRWGYQWNP4DLIHRQI/graph.json","fetch_events":"https://pith.science/api/pith-number/5LL2BD22YNRWGYQWNP4DLIHRQI/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/5LL2BD22YNRWGYQWNP4DLIHRQI/action/timestamp_anchor","attest_storage":"https://pith.science/pith/5LL2BD22YNRWGYQWNP4DLIHRQI/action/storage_attestation","attest_author":"https://pith.science/pith/5LL2BD22YNRWGYQWNP4DLIHRQI/action/author_attestation","sign_citation":"https://pith.science/pith/5LL2BD22YNRWGYQWNP4DLIHRQI/action/citation_signature","submit_replication":"https://pith.science/pith/5LL2BD22YNRWGYQWNP4DLIHRQI/action/replication_record"}},"created_at":"2026-07-05T05:27:05.780002+00:00","updated_at":"2026-07-05T05:27:05.780002+00:00"}