{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:AHGV3LSGSENHPJ66SKS4CQHEOI","short_pith_number":"pith:AHGV3LSG","schema_version":"1.0","canonical_sha256":"01cd5dae46911a77a7de92a5c140e47232233f775cd8acd1129739c6301a8455","source":{"kind":"arxiv","id":"2311.07989","version":7},"attestation_state":"computed","paper":{"title":"Unifying the Perspectives of NLP and Software Engineering: A Survey on Language Models for Code","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.AI","cs.SE"],"primary_cat":"cs.CL","authors_text":"Bingchang Liu, Chaoyu Chen, Cong Liao, Hang Yu, Jianguo Li, Rui Wang, Zi Gong, Ziyin Zhang","submitted_at":"2023-11-14T08:34:26Z","abstract_excerpt":"In this work we systematically review the recent advancements in software engineering with language models, covering 70+ models, 40+ evaluation tasks, 180+ datasets, and 900 related works. Unlike previous works, we integrate software engineering (SE) with natural language processing (NLP) by discussing the perspectives of both sides: SE applies language models for development automation, while NLP adopts SE tasks for language model evaluation. We break down code processing models into general language models represented by the GPT family and specialized models that are specifically pretrained "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2311.07989","kind":"arxiv","version":7},"metadata":{"license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","primary_cat":"cs.CL","submitted_at":"2023-11-14T08:34:26Z","cross_cats_sorted":["cs.AI","cs.SE"],"title_canon_sha256":"a010262a810342dd8f198391f59640885418c88afaa47f95a2818422f2b714ca","abstract_canon_sha256":"2e061cbdd27a904682c3b07456da166fe2091b4fc0e37161142bdec4eb39c629"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T08:36:44.301108Z","signature_b64":"Div9VA7zpX4SjIH43yIn/VZh58L05qOGNbdpYpSHG3Bq3bH075IZ1MOY6E9aTa5FSeEhsmnA3x3G3WAi9iAxBQ==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"01cd5dae46911a77a7de92a5c140e47232233f775cd8acd1129739c6301a8455","last_reissued_at":"2026-07-05T08:36:44.300634Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T08:36:44.300634Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Unifying the Perspectives of NLP and Software Engineering: A Survey on Language Models for Code","license":"http://creativecommons.org/licenses/by-nc-nd/4.0/","headline":"","cross_cats":["cs.AI","cs.SE"],"primary_cat":"cs.CL","authors_text":"Bingchang Liu, Chaoyu Chen, Cong Liao, Hang Yu, Jianguo Li, Rui Wang, Zi Gong, Ziyin Zhang","submitted_at":"2023-11-14T08:34:26Z","abstract_excerpt":"In this work we systematically review the recent advancements in software engineering with language models, covering 70+ models, 40+ evaluation tasks, 180+ datasets, and 900 related works. Unlike previous works, we integrate software engineering (SE) with natural language processing (NLP) by discussing the perspectives of both sides: SE applies language models for development automation, while NLP adopts SE tasks for language model evaluation. We break down code processing models into general language models represented by the GPT family and specialized models that are specifically pretrained "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2311.07989","kind":"arxiv","version":7},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2311.07989/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2311.07989","created_at":"2026-07-05T08:36:44.300689+00:00"},{"alias_kind":"arxiv_version","alias_value":"2311.07989v7","created_at":"2026-07-05T08:36:44.300689+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2311.07989","created_at":"2026-07-05T08:36:44.300689+00:00"},{"alias_kind":"pith_short_12","alias_value":"AHGV3LSGSENH","created_at":"2026-07-05T08:36:44.300689+00:00"},{"alias_kind":"pith_short_16","alias_value":"AHGV3LSGSENHPJ66","created_at":"2026-07-05T08:36:44.300689+00:00"},{"alias_kind":"pith_short_8","alias_value":"AHGV3LSG","created_at":"2026-07-05T08:36:44.300689+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":9,"internal_anchor_count":0,"sample":[{"citing_arxiv_id":"2606.10106","citing_title":"What makes a harness a harness: necessary and sufficient conditions for an agent harness","ref_index":66,"is_internal_anchor":false},{"citing_arxiv_id":"2502.14925","citing_title":"CODEPROMPTZIP: Code-specific Prompt Compression for Retrieval-Augmented Generation in Coding Tasks with LMs","ref_index":15,"is_internal_anchor":false},{"citing_arxiv_id":"2503.17181","citing_title":"A Study of LLMs' Preferences for Libraries and Programming Languages","ref_index":86,"is_internal_anchor":false},{"citing_arxiv_id":"2605.17978","citing_title":"AutoVecCoder: Teaching LLMs to Generate Explicitly Vectorized Code","ref_index":96,"is_internal_anchor":false},{"citing_arxiv_id":"2605.19102","citing_title":"Prompt Optimization for LLM Code Generation via Reinforcement Learning","ref_index":36,"is_internal_anchor":false},{"citing_arxiv_id":"2602.01785","citing_title":"CodeOCR: On the Effectiveness of Vision Language Models in Code Understanding","ref_index":109,"is_internal_anchor":false},{"citing_arxiv_id":"2604.26523","citing_title":"RepoDoc: A Knowledge Graph-Based Framework to Automatic Documentation Generation and Incremental Updates","ref_index":38,"is_internal_anchor":false},{"citing_arxiv_id":"2604.07769","citing_title":"An Empirical Study on Influence-Based Pretraining Data Selection for Code Large Language Models","ref_index":49,"is_internal_anchor":false},{"citing_arxiv_id":"2504.15564","citing_title":"OpenClassGen: A Large-Scale Corpus of Real-World Python Classes for LLM Research","ref_index":59,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/AHGV3LSGSENHPJ66SKS4CQHEOI","json":"https://pith.science/pith/AHGV3LSGSENHPJ66SKS4CQHEOI.json","graph_json":"https://pith.science/api/pith-number/AHGV3LSGSENHPJ66SKS4CQHEOI/graph.json","events_json":"https://pith.science/api/pith-number/AHGV3LSGSENHPJ66SKS4CQHEOI/events.json","paper":"https://pith.science/paper/AHGV3LSG"},"agent_actions":{"view_html":"https://pith.science/pith/AHGV3LSGSENHPJ66SKS4CQHEOI","download_json":"https://pith.science/pith/AHGV3LSGSENHPJ66SKS4CQHEOI.json","view_paper":"https://pith.science/paper/AHGV3LSG","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2311.07989&json=true","fetch_graph":"https://pith.science/api/pith-number/AHGV3LSGSENHPJ66SKS4CQHEOI/graph.json","fetch_events":"https://pith.science/api/pith-number/AHGV3LSGSENHPJ66SKS4CQHEOI/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/AHGV3LSGSENHPJ66SKS4CQHEOI/action/timestamp_anchor","attest_storage":"https://pith.science/pith/AHGV3LSGSENHPJ66SKS4CQHEOI/action/storage_attestation","attest_author":"https://pith.science/pith/AHGV3LSGSENHPJ66SKS4CQHEOI/action/author_attestation","sign_citation":"https://pith.science/pith/AHGV3LSGSENHPJ66SKS4CQHEOI/action/citation_signature","submit_replication":"https://pith.science/pith/AHGV3LSGSENHPJ66SKS4CQHEOI/action/replication_record"}},"created_at":"2026-07-05T08:36:44.300689+00:00","updated_at":"2026-07-05T08:36:44.300689+00:00"}