{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:SPAYMY2YJDTZZVOATSJJXBKG3F","short_pith_number":"pith:SPAYMY2Y","schema_version":"1.0","canonical_sha256":"93c186635848e79cd5c09c929b8546d9513b4809aef4fcd2c672235e2c22bd87","source":{"kind":"arxiv","id":"2302.05020","version":3},"attestation_state":"computed","paper":{"title":"Impact of Code Language Models on Automated Program Repair","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.SE","authors_text":"Kevin Liu, Lin Tan, Nan Jiang, Thibaud Lutellier","submitted_at":"2023-02-10T02:29:37Z","abstract_excerpt":"Automated program repair (APR) aims to help developers improve software reliability by generating patches for buggy programs. Although many code language models (CLM) are developed and effective in many software tasks such as code completion, there has been little comprehensive, in-depth work to evaluate CLMs' fixing capabilities and to fine-tune CLMs for the APR task.\n  Firstly, this work is the first to evaluate ten CLMs on four APR benchmarks, which shows that surprisingly, the best CLM, as is, fixes 72% more bugs than the state-of-the-art deep-learning (DL)-based APR techniques. Secondly, "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2302.05020","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.SE","submitted_at":"2023-02-10T02:29:37Z","cross_cats_sorted":[],"title_canon_sha256":"f66a36abf71dfe3869d6493e4b609ff5c4111c996658c26cabae3ed4b9ac4aa7","abstract_canon_sha256":"ce76c3d31ffa945c9dd38d865bef373071905dae9c6ffc25b75236d0a65c1020"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:01:17.139825Z","signature_b64":"mvXJt7RgNb0LzORM+f3sUkmJyiawmxNyFeSiAzQw244wChjCiAJfTesvceOrReRW61ArE7FBfH7yPMPESsCQCw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"93c186635848e79cd5c09c929b8546d9513b4809aef4fcd2c672235e2c22bd87","last_reissued_at":"2026-07-05T06:01:17.139343Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:01:17.139343Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Impact of Code Language Models on Automated Program Repair","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.SE","authors_text":"Kevin Liu, Lin Tan, Nan Jiang, Thibaud Lutellier","submitted_at":"2023-02-10T02:29:37Z","abstract_excerpt":"Automated program repair (APR) aims to help developers improve software reliability by generating patches for buggy programs. Although many code language models (CLM) are developed and effective in many software tasks such as code completion, there has been little comprehensive, in-depth work to evaluate CLMs' fixing capabilities and to fine-tune CLMs for the APR task.\n  Firstly, this work is the first to evaluate ten CLMs on four APR benchmarks, which shows that surprisingly, the best CLM, as is, fixes 72% more bugs than the state-of-the-art deep-learning (DL)-based APR techniques. Secondly, "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2302.05020","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2302.05020/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2302.05020","created_at":"2026-07-05T06:01:17.139402+00:00"},{"alias_kind":"arxiv_version","alias_value":"2302.05020v3","created_at":"2026-07-05T06:01:17.139402+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2302.05020","created_at":"2026-07-05T06:01:17.139402+00:00"},{"alias_kind":"pith_short_12","alias_value":"SPAYMY2YJDTZ","created_at":"2026-07-05T06:01:17.139402+00:00"},{"alias_kind":"pith_short_16","alias_value":"SPAYMY2YJDTZZVOA","created_at":"2026-07-05T06:01:17.139402+00:00"},{"alias_kind":"pith_short_8","alias_value":"SPAYMY2Y","created_at":"2026-07-05T06:01:17.139402+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2401.16310","citing_title":"An Insight into Security Code Review with LLMs: Capabilities, Obstacles, and Influential Factors","ref_index":2,"is_internal_anchor":false},{"citing_arxiv_id":"2305.01210","citing_title":"Is Your Code Generated by ChatGPT Really Correct? Rigorous Evaluation of Large Language Models for Code Generation","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2403.07974","citing_title":"LiveCodeBench: Holistic and Contamination Free Evaluation of Large Language Models for Code","ref_index":72,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/SPAYMY2YJDTZZVOATSJJXBKG3F","json":"https://pith.science/pith/SPAYMY2YJDTZZVOATSJJXBKG3F.json","graph_json":"https://pith.science/api/pith-number/SPAYMY2YJDTZZVOATSJJXBKG3F/graph.json","events_json":"https://pith.science/api/pith-number/SPAYMY2YJDTZZVOATSJJXBKG3F/events.json","paper":"https://pith.science/paper/SPAYMY2Y"},"agent_actions":{"view_html":"https://pith.science/pith/SPAYMY2YJDTZZVOATSJJXBKG3F","download_json":"https://pith.science/pith/SPAYMY2YJDTZZVOATSJJXBKG3F.json","view_paper":"https://pith.science/paper/SPAYMY2Y","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2302.05020&json=true","fetch_graph":"https://pith.science/api/pith-number/SPAYMY2YJDTZZVOATSJJXBKG3F/graph.json","fetch_events":"https://pith.science/api/pith-number/SPAYMY2YJDTZZVOATSJJXBKG3F/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/SPAYMY2YJDTZZVOATSJJXBKG3F/action/timestamp_anchor","attest_storage":"https://pith.science/pith/SPAYMY2YJDTZZVOATSJJXBKG3F/action/storage_attestation","attest_author":"https://pith.science/pith/SPAYMY2YJDTZZVOATSJJXBKG3F/action/author_attestation","sign_citation":"https://pith.science/pith/SPAYMY2YJDTZZVOATSJJXBKG3F/action/citation_signature","submit_replication":"https://pith.science/pith/SPAYMY2YJDTZZVOATSJJXBKG3F/action/replication_record"}},"created_at":"2026-07-05T06:01:17.139402+00:00","updated_at":"2026-07-05T06:01:17.139402+00:00"}