{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:FN62Q2VLDXZTNWZT76JFKOGFZF","short_pith_number":"pith:FN62Q2VL","schema_version":"1.0","canonical_sha256":"2b7da86aab1df336db33ff925538c5c967eb9e41c3c7a4ced5461559c1215696","source":{"kind":"arxiv","id":"2305.15507","version":1},"attestation_state":"computed","paper":{"title":"The Larger They Are, the Harder They Fail: Language Models do not Recognize Identifier Swaps in Python","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Antonio Valerio Miceli-Barone, Fazl Barez, Ioannis Konstas, Shay B. Cohen","submitted_at":"2023-05-24T18:54:39Z","abstract_excerpt":"Large Language Models (LLMs) have successfully been applied to code generation tasks, raising the question of how well these models understand programming. Typical programming languages have invariances and equivariances in their semantics that human programmers intuitively understand and exploit, such as the (near) invariance to the renaming of identifiers. We show that LLMs not only fail to properly generate correct Python code when default function names are swapped, but some of them even become more confident in their incorrect predictions as the model size increases, an instance of the re"},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2305.15507","kind":"arxiv","version":1},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.CL","submitted_at":"2023-05-24T18:54:39Z","cross_cats_sorted":["cs.AI"],"title_canon_sha256":"adc2c8a11542a64dbd14d89c7305d0a9ed64c6d1b9d861960fcacbcf275ad461","abstract_canon_sha256":"f387b3489cae9f30ed60f6146a59d8d7fd1e3550ccdb17959c21a6cc77180094"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T06:13:41.299286Z","signature_b64":"ylWm+xOCgenmBMfOLUtJ9wAFHV8fu3M4IRuE8MnF6S24CYtspKqSjz8E1aFcNJh3EHRjCG3XPqaul8vl5tpgBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"2b7da86aab1df336db33ff925538c5c967eb9e41c3c7a4ced5461559c1215696","last_reissued_at":"2026-07-05T06:13:41.298808Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T06:13:41.298808Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"The Larger They Are, the Harder They Fail: Language Models do not Recognize Identifier Swaps in Python","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":["cs.AI"],"primary_cat":"cs.CL","authors_text":"Antonio Valerio Miceli-Barone, Fazl Barez, Ioannis Konstas, Shay B. Cohen","submitted_at":"2023-05-24T18:54:39Z","abstract_excerpt":"Large Language Models (LLMs) have successfully been applied to code generation tasks, raising the question of how well these models understand programming. Typical programming languages have invariances and equivariances in their semantics that human programmers intuitively understand and exploit, such as the (near) invariance to the renaming of identifiers. We show that LLMs not only fail to properly generate correct Python code when default function names are swapped, but some of them even become more confident in their incorrect predictions as the model size increases, an instance of the re"},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2305.15507","kind":"arxiv","version":1},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2305.15507/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2305.15507","created_at":"2026-07-05T06:13:41.298865+00:00"},{"alias_kind":"arxiv_version","alias_value":"2305.15507v1","created_at":"2026-07-05T06:13:41.298865+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2305.15507","created_at":"2026-07-05T06:13:41.298865+00:00"},{"alias_kind":"pith_short_12","alias_value":"FN62Q2VLDXZT","created_at":"2026-07-05T06:13:41.298865+00:00"},{"alias_kind":"pith_short_16","alias_value":"FN62Q2VLDXZTNWZT","created_at":"2026-07-05T06:13:41.298865+00:00"},{"alias_kind":"pith_short_8","alias_value":"FN62Q2VL","created_at":"2026-07-05T06:13:41.298865+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":3,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2402.09664","citing_title":"CodeMind: Evaluating Large Language Models for Code Reasoning","ref_index":7,"is_internal_anchor":false},{"citing_arxiv_id":"2510.15079","citing_title":"Assessing Coherency and Consistency of Code Execution Reasoning by Large Language Models","ref_index":33,"is_internal_anchor":false},{"citing_arxiv_id":"2403.07974","citing_title":"LiveCodeBench: Holistic and Contamination Free Evaluation of Large Language Models for Code","ref_index":87,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/FN62Q2VLDXZTNWZT76JFKOGFZF","json":"https://pith.science/pith/FN62Q2VLDXZTNWZT76JFKOGFZF.json","graph_json":"https://pith.science/api/pith-number/FN62Q2VLDXZTNWZT76JFKOGFZF/graph.json","events_json":"https://pith.science/api/pith-number/FN62Q2VLDXZTNWZT76JFKOGFZF/events.json","paper":"https://pith.science/paper/FN62Q2VL"},"agent_actions":{"view_html":"https://pith.science/pith/FN62Q2VLDXZTNWZT76JFKOGFZF","download_json":"https://pith.science/pith/FN62Q2VLDXZTNWZT76JFKOGFZF.json","view_paper":"https://pith.science/paper/FN62Q2VL","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2305.15507&json=true","fetch_graph":"https://pith.science/api/pith-number/FN62Q2VLDXZTNWZT76JFKOGFZF/graph.json","fetch_events":"https://pith.science/api/pith-number/FN62Q2VLDXZTNWZT76JFKOGFZF/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/FN62Q2VLDXZTNWZT76JFKOGFZF/action/timestamp_anchor","attest_storage":"https://pith.science/pith/FN62Q2VLDXZTNWZT76JFKOGFZF/action/storage_attestation","attest_author":"https://pith.science/pith/FN62Q2VLDXZTNWZT76JFKOGFZF/action/author_attestation","sign_citation":"https://pith.science/pith/FN62Q2VLDXZTNWZT76JFKOGFZF/action/citation_signature","submit_replication":"https://pith.science/pith/FN62Q2VLDXZTNWZT76JFKOGFZF/action/replication_record"}},"created_at":"2026-07-05T06:13:41.298865+00:00","updated_at":"2026-07-05T06:13:41.298865+00:00"}