{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:TEJKS4WZZIO2NFOBKIIXR6KTDQ","short_pith_number":"pith:TEJKS4WZ","schema_version":"1.0","canonical_sha256":"9912a972d9ca1da695c1521178f9531c1ea094dc2e02b41b8203bc638a9d431d","source":{"kind":"arxiv","id":"2404.03543","version":3},"attestation_state":"computed","paper":{"title":"CodeEditorBench: Evaluating Code Editing Capability of Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.LG"],"primary_cat":"cs.SE","authors_text":"Ding Pan, Ge Zhang, Jiawei Guo, Jie Fu, Kaijing Ma, Ruibo Liu, Shuyue Guo, Tianyu Zheng, Wenhu Chen, Xiang Yue, Xingwei Qu, Xueling Liu, Yizhi Li, Yue Wang, Zhouliang Yu, Ziming Li","submitted_at":"2024-04-04T15:49:49Z","abstract_excerpt":"Large Language Models (LLMs) for code are rapidly evolving, with code editing emerging as a critical capability. We introduce CodeEditorBench, an evaluation framework designed to rigorously assess the performance of LLMs in code editing tasks, including debugging, translating, polishing, and requirement switching. Unlike existing benchmarks focusing solely on code generation, CodeEditorBench emphasizes real-world scenarios and practical aspects of software development. We curate diverse coding challenges and scenarios from five sources, covering various programming languages, complexity levels"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2404.03543","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.SE","submitted_at":"2024-04-04T15:49:49Z","cross_cats_sorted":["cs.AI","cs.CL","cs.LG"],"title_canon_sha256":"dcf40fee9872ab2856667e6f6276f0fb6340ce3222f1c9daa19cc611f9fe14ad","abstract_canon_sha256":"0cb266f228c665ef829ca34c214603ab60b1428e4f27cdb2fe69ddadcf87cd2e"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:45:51.745063Z","signature_b64":"jx52sicmBbiGJMLqx2iheXci+q/coiE06poPNNuThpKBpSC/ABCbh2FgIq8mFC3zLjE/DWbglauUlRPsfrh0Dg==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"9912a972d9ca1da695c1521178f9531c1ea094dc2e02b41b8203bc638a9d431d","last_reissued_at":"2026-07-05T10:45:51.744474Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:45:51.744474Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"CodeEditorBench: Evaluating Code Editing Capability of Large Language Models","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.AI","cs.CL","cs.LG"],"primary_cat":"cs.SE","authors_text":"Ding Pan, Ge Zhang, Jiawei Guo, Jie Fu, Kaijing Ma, Ruibo Liu, Shuyue Guo, Tianyu Zheng, Wenhu Chen, Xiang Yue, Xingwei Qu, Xueling Liu, Yizhi Li, Yue Wang, Zhouliang Yu, Ziming Li","submitted_at":"2024-04-04T15:49:49Z","abstract_excerpt":"Large Language Models (LLMs) for code are rapidly evolving, with code editing emerging as a critical capability. We introduce CodeEditorBench, an evaluation framework designed to rigorously assess the performance of LLMs in code editing tasks, including debugging, translating, polishing, and requirement switching. Unlike existing benchmarks focusing solely on code generation, CodeEditorBench emphasizes real-world scenarios and practical aspects of software development. We curate diverse coding challenges and scenarios from five sources, covering various programming languages, complexity levels"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2404.03543","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2404.03543/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2404.03543","created_at":"2026-07-05T10:45:51.744544+00:00"},{"alias_kind":"arxiv_version","alias_value":"2404.03543v3","created_at":"2026-07-05T10:45:51.744544+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2404.03543","created_at":"2026-07-05T10:45:51.744544+00:00"},{"alias_kind":"pith_short_12","alias_value":"TEJKS4WZZIO2","created_at":"2026-07-05T10:45:51.744544+00:00"},{"alias_kind":"pith_short_16","alias_value":"TEJKS4WZZIO2NFOB","created_at":"2026-07-05T10:45:51.744544+00:00"},{"alias_kind":"pith_short_8","alias_value":"TEJKS4WZ","created_at":"2026-07-05T10:45:51.744544+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.05249","citing_title":"SWE-InfraBench: Evaluating Language Models on Cloud Infrastructure Code","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2604.25737","citing_title":"SAFEdit: Does Multi-Agent Decomposition Resolve the Reliability Challenges of Instructed Code Editing?","ref_index":8,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06068","citing_title":"VibeServe: Can AI Agents Build Bespoke LLM Serving Systems?","ref_index":21,"is_internal_anchor":false},{"citing_arxiv_id":"2604.15597","citing_title":"LLMs Corrupt Your Documents When You Delegate","ref_index":24,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/TEJKS4WZZIO2NFOBKIIXR6KTDQ","json":"https://pith.science/pith/TEJKS4WZZIO2NFOBKIIXR6KTDQ.json","graph_json":"https://pith.science/api/pith-number/TEJKS4WZZIO2NFOBKIIXR6KTDQ/graph.json","events_json":"https://pith.science/api/pith-number/TEJKS4WZZIO2NFOBKIIXR6KTDQ/events.json","paper":"https://pith.science/paper/TEJKS4WZ"},"agent_actions":{"view_html":"https://pith.science/pith/TEJKS4WZZIO2NFOBKIIXR6KTDQ","download_json":"https://pith.science/pith/TEJKS4WZZIO2NFOBKIIXR6KTDQ.json","view_paper":"https://pith.science/paper/TEJKS4WZ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2404.03543&json=true","fetch_graph":"https://pith.science/api/pith-number/TEJKS4WZZIO2NFOBKIIXR6KTDQ/graph.json","fetch_events":"https://pith.science/api/pith-number/TEJKS4WZZIO2NFOBKIIXR6KTDQ/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/TEJKS4WZZIO2NFOBKIIXR6KTDQ/action/timestamp_anchor","attest_storage":"https://pith.science/pith/TEJKS4WZZIO2NFOBKIIXR6KTDQ/action/storage_attestation","attest_author":"https://pith.science/pith/TEJKS4WZZIO2NFOBKIIXR6KTDQ/action/author_attestation","sign_citation":"https://pith.science/pith/TEJKS4WZZIO2NFOBKIIXR6KTDQ/action/citation_signature","submit_replication":"https://pith.science/pith/TEJKS4WZZIO2NFOBKIIXR6KTDQ/action/replication_record"}},"created_at":"2026-07-05T10:45:51.744544+00:00","updated_at":"2026-07-05T10:45:51.744544+00:00"}