{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2024:SCX6FV4EBSZ6D2RJAIBQMXK2HD","short_pith_number":"pith:SCX6FV4E","schema_version":"1.0","canonical_sha256":"90afe2d7840cb3e1ea290203065d5a38f40b15ea1056571e998cd10d518d4a92","source":{"kind":"arxiv","id":"2407.06249","version":3},"attestation_state":"computed","paper":{"title":"CodeUpdateArena: Benchmarking Knowledge Editing on API Updates","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.SE"],"primary_cat":"cs.CL","authors_text":"Eunsol Choi, Greg Durrett, Shrey Pandit, Xi Ye, Zeyu Leo Liu","submitted_at":"2024-07-08T17:55:04Z","abstract_excerpt":"Large language models (LLMs) are increasingly being used to synthesize and reason about source code. However, the static nature of these models' knowledge does not reflect the fact that libraries and API functions they invoke are continuously evolving, with functionality being added or changing. While numerous benchmarks evaluate how LLMs can generate code, no prior work has studied how an LLMs' knowledge about code API functions can be updated. To fill this gap, we present CodeUpdateArena, a benchmark for knowledge editing in the code domain. An instance in our benchmark consists of a synthet"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2407.06249","kind":"arxiv","version":3},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2024-07-08T17:55:04Z","cross_cats_sorted":["cs.SE"],"title_canon_sha256":"90e93752392846366c8cb03bb7c4b43f289b77939282d77b71084038d2de81fd","abstract_canon_sha256":"ff61349877992d34b3bed4dd593aa89c276454cfaca39290c88087754e2483d9"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T10:43:36.630064Z","signature_b64":"vBsDq7WpYiiC9HePaBPV0DWl3KEMDCQVzzjlMXiMb3L0nmBqVtOkpqKf3i13vIlnq6wXi9iLqvRiWO9X32XVBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"90afe2d7840cb3e1ea290203065d5a38f40b15ea1056571e998cd10d518d4a92","last_reissued_at":"2026-07-05T10:43:36.629610Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T10:43:36.629610Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"CodeUpdateArena: Benchmarking Knowledge Editing on API Updates","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.SE"],"primary_cat":"cs.CL","authors_text":"Eunsol Choi, Greg Durrett, Shrey Pandit, Xi Ye, Zeyu Leo Liu","submitted_at":"2024-07-08T17:55:04Z","abstract_excerpt":"Large language models (LLMs) are increasingly being used to synthesize and reason about source code. However, the static nature of these models' knowledge does not reflect the fact that libraries and API functions they invoke are continuously evolving, with functionality being added or changing. While numerous benchmarks evaluate how LLMs can generate code, no prior work has studied how an LLMs' knowledge about code API functions can be updated. To fill this gap, we present CodeUpdateArena, a benchmark for knowledge editing in the code domain. An instance in our benchmark consists of a synthet"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2407.06249","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2407.06249/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2407.06249","created_at":"2026-07-05T10:43:36.629668+00:00"},{"alias_kind":"arxiv_version","alias_value":"2407.06249v3","created_at":"2026-07-05T10:43:36.629668+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2407.06249","created_at":"2026-07-05T10:43:36.629668+00:00"},{"alias_kind":"pith_short_12","alias_value":"SCX6FV4EBSZ6","created_at":"2026-07-05T10:43:36.629668+00:00"},{"alias_kind":"pith_short_16","alias_value":"SCX6FV4EBSZ6D2RJ","created_at":"2026-07-05T10:43:36.629668+00:00"},{"alias_kind":"pith_short_8","alias_value":"SCX6FV4E","created_at":"2026-07-05T10:43:36.629668+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":4,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.05238","citing_title":"DeployBench: Benchmarking LLM Agents for Research Artifact Deployment","ref_index":19,"is_internal_anchor":false},{"citing_arxiv_id":"2605.31478","citing_title":"Knowledge Boundary Probing and Demand-Guided Intervention for LLM-Based Power System Code Generation","ref_index":9,"is_internal_anchor":false},{"citing_arxiv_id":"2511.03182","citing_title":"Understanding Robustness of Model Editing in Code LLMs","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2605.06279","citing_title":"Correct Code, Vulnerable Dependencies: A Large Scale Measurement Study of LLM-Specified Library Versions","ref_index":39,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/SCX6FV4EBSZ6D2RJAIBQMXK2HD","json":"https://pith.science/pith/SCX6FV4EBSZ6D2RJAIBQMXK2HD.json","graph_json":"https://pith.science/api/pith-number/SCX6FV4EBSZ6D2RJAIBQMXK2HD/graph.json","events_json":"https://pith.science/api/pith-number/SCX6FV4EBSZ6D2RJAIBQMXK2HD/events.json","paper":"https://pith.science/paper/SCX6FV4E"},"agent_actions":{"view_html":"https://pith.science/pith/SCX6FV4EBSZ6D2RJAIBQMXK2HD","download_json":"https://pith.science/pith/SCX6FV4EBSZ6D2RJAIBQMXK2HD.json","view_paper":"https://pith.science/paper/SCX6FV4E","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2407.06249&json=true","fetch_graph":"https://pith.science/api/pith-number/SCX6FV4EBSZ6D2RJAIBQMXK2HD/graph.json","fetch_events":"https://pith.science/api/pith-number/SCX6FV4EBSZ6D2RJAIBQMXK2HD/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/SCX6FV4EBSZ6D2RJAIBQMXK2HD/action/timestamp_anchor","attest_storage":"https://pith.science/pith/SCX6FV4EBSZ6D2RJAIBQMXK2HD/action/storage_attestation","attest_author":"https://pith.science/pith/SCX6FV4EBSZ6D2RJAIBQMXK2HD/action/author_attestation","sign_citation":"https://pith.science/pith/SCX6FV4EBSZ6D2RJAIBQMXK2HD/action/citation_signature","submit_replication":"https://pith.science/pith/SCX6FV4EBSZ6D2RJAIBQMXK2HD/action/replication_record"}},"created_at":"2026-07-05T10:43:36.629668+00:00","updated_at":"2026-07-05T10:43:36.629668+00:00"}