{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2022:CVICRCRJDSTGWJIPQNQM77ATLO","short_pith_number":"pith:CVICRCRJ","schema_version":"1.0","canonical_sha256":"1550288a291ca66b250f8360cffc135b94a8b416f91f83186d3199a8965a8864","source":{"kind":"arxiv","id":"2202.13169","version":3},"attestation_state":"computed","paper":{"title":"A Systematic Evaluation of Large Language Models of Code","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.PL","authors_text":"Frank F. Xu, Graham Neubig, Uri Alon, Vincent J. Hellendoorn","submitted_at":"2022-02-26T15:53:55Z","abstract_excerpt":"Large language models (LMs) of code have recently shown tremendous promise in completing code and synthesizing code from natural language descriptions. However, the current state-of-the-art code LMs (e.g., Codex (Chen et al., 2021)) are not publicly available, leaving many questions about their model and data design decisions. We aim to fill in some of these blanks through a systematic evaluation of the largest existing models: Codex, GPT-J, GPT-Neo, GPT-NeoX-20B, and CodeParrot, across various programming languages. Although Codex itself is not open-source, we find that existing open-source m"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2202.13169","kind":"arxiv","version":3},"metadata":{"license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","primary_cat":"cs.PL","submitted_at":"2022-02-26T15:53:55Z","cross_cats_sorted":["cs.CL"],"title_canon_sha256":"2fa9cd6494c03254530709cdaf5f3a7f197119e03659abfb7cc6bb4a4082ddc1","abstract_canon_sha256":"ccd5d8b1e9eeeb02d7fd211af95d006ddf84b1287d18e9a09e76ee716af8f959"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T04:20:18.712898Z","signature_b64":"4g807EzEdtw2vAtHBweAAYXud4a+DfMeutGgxyHomq9PhLxXA4vdT/JWi4UIZbLP+VF0CkGwom+URDZVwulJBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"1550288a291ca66b250f8360cffc135b94a8b416f91f83186d3199a8965a8864","last_reissued_at":"2026-07-05T04:20:18.712428Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T04:20:18.712428Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"A Systematic Evaluation of Large Language Models of Code","license":"http://arxiv.org/licenses/nonexclusive-distrib/1.0/","headline":"","cross_cats":["cs.CL"],"primary_cat":"cs.PL","authors_text":"Frank F. Xu, Graham Neubig, Uri Alon, Vincent J. Hellendoorn","submitted_at":"2022-02-26T15:53:55Z","abstract_excerpt":"Large language models (LMs) of code have recently shown tremendous promise in completing code and synthesizing code from natural language descriptions. However, the current state-of-the-art code LMs (e.g., Codex (Chen et al., 2021)) are not publicly available, leaving many questions about their model and data design decisions. We aim to fill in some of these blanks through a systematic evaluation of the largest existing models: Codex, GPT-J, GPT-Neo, GPT-NeoX-20B, and CodeParrot, across various programming languages. Although Codex itself is not open-source, we find that existing open-source m"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2202.13169","kind":"arxiv","version":3},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2202.13169/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2202.13169","created_at":"2026-07-05T04:20:18.712493+00:00"},{"alias_kind":"arxiv_version","alias_value":"2202.13169v3","created_at":"2026-07-05T04:20:18.712493+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2202.13169","created_at":"2026-07-05T04:20:18.712493+00:00"},{"alias_kind":"pith_short_12","alias_value":"CVICRCRJDSTG","created_at":"2026-07-05T04:20:18.712493+00:00"},{"alias_kind":"pith_short_16","alias_value":"CVICRCRJDSTGWJIP","created_at":"2026-07-05T04:20:18.712493+00:00"},{"alias_kind":"pith_short_8","alias_value":"CVICRCRJ","created_at":"2026-07-05T04:20:18.712493+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":6,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2505.13766","citing_title":"A Blueprint for AI-Driven Software Quality: Integrating LLMs with Established Standards","ref_index":23,"is_internal_anchor":false},{"citing_arxiv_id":"2405.14782","citing_title":"Lessons from the Trenches on Reproducible Evaluation of Language Models","ref_index":75,"is_internal_anchor":false},{"citing_arxiv_id":"2204.05999","citing_title":"InCoder: A Generative Model for Code Infilling and Synthesis","ref_index":26,"is_internal_anchor":false},{"citing_arxiv_id":"2306.03091","citing_title":"RepoBench: Benchmarking Repository-Level Code Auto-Completion Systems","ref_index":48,"is_internal_anchor":false},{"citing_arxiv_id":"2304.01373","citing_title":"Pythia: A Suite for Analyzing Large Language Models Across Training and Scaling","ref_index":175,"is_internal_anchor":false},{"citing_arxiv_id":"2603.26567","citing_title":"Beyond Code Snippets: Benchmarking LLMs on Repository-Level Question Answering","ref_index":59,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/CVICRCRJDSTGWJIPQNQM77ATLO","json":"https://pith.science/pith/CVICRCRJDSTGWJIPQNQM77ATLO.json","graph_json":"https://pith.science/api/pith-number/CVICRCRJDSTGWJIPQNQM77ATLO/graph.json","events_json":"https://pith.science/api/pith-number/CVICRCRJDSTGWJIPQNQM77ATLO/events.json","paper":"https://pith.science/paper/CVICRCRJ"},"agent_actions":{"view_html":"https://pith.science/pith/CVICRCRJDSTGWJIPQNQM77ATLO","download_json":"https://pith.science/pith/CVICRCRJDSTGWJIPQNQM77ATLO.json","view_paper":"https://pith.science/paper/CVICRCRJ","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2202.13169&json=true","fetch_graph":"https://pith.science/api/pith-number/CVICRCRJDSTGWJIPQNQM77ATLO/graph.json","fetch_events":"https://pith.science/api/pith-number/CVICRCRJDSTGWJIPQNQM77ATLO/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/CVICRCRJDSTGWJIPQNQM77ATLO/action/timestamp_anchor","attest_storage":"https://pith.science/pith/CVICRCRJDSTGWJIPQNQM77ATLO/action/storage_attestation","attest_author":"https://pith.science/pith/CVICRCRJDSTGWJIPQNQM77ATLO/action/author_attestation","sign_citation":"https://pith.science/pith/CVICRCRJDSTGWJIPQNQM77ATLO/action/citation_signature","submit_replication":"https://pith.science/pith/CVICRCRJDSTGWJIPQNQM77ATLO/action/replication_record"}},"created_at":"2026-07-05T04:20:18.712493+00:00","updated_at":"2026-07-05T04:20:18.712493+00:00"}