{"record_type":"pith_number_record","schema_url":"https://pith.science/schemas/pith-number/v1.json","pith_number":"pith:2023:UJIZDJM3AYGALKXOETGIL4OTM3","short_pith_number":"pith:UJIZDJM3","schema_version":"1.0","canonical_sha256":"a25191a59b060c05aaee24cc85f1d366fc6a17ddc374bb9a02c13e3fa16f9a6e","source":{"kind":"arxiv","id":"2310.03533","version":4},"attestation_state":"computed","paper":{"title":"Large Language Models for Software Engineering: Survey and Open Problems","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.SE","authors_text":"Angela Fan, Beliz Gokkaya, Jie M. Zhang, Mark Harman, Mitya Lyubarskiy, Shin Yoo, Shubho Sengupta","submitted_at":"2023-10-05T13:33:26Z","abstract_excerpt":"This paper provides a survey of the emerging area of Large Language Models (LLMs) for Software Engineering (SE). It also sets out open research challenges for the application of LLMs to technical problems faced by software engineers. LLMs' emergent properties bring novelty and creativity with applications right across the spectrum of Software Engineering activities including coding, design, requirements, repair, refactoring, performance improvement, documentation and analytics. However, these very same emergent properties also pose significant technical challenges; we need techniques that can "},"verification_status":{"content_addressed":true,"pith_receipt":true,"author_attested":false,"weak_author_claims":0,"strong_author_claims":0,"externally_anchored":false,"storage_verified":false,"citation_signatures":0,"replication_records":0,"graph_snapshot":true,"references_resolved":false,"formal_links_present":false},"canonical_record":{"source":{"id":"2310.03533","kind":"arxiv","version":4},"metadata":{"license":"http://creativecommons.org/licenses/by/4.0/","primary_cat":"cs.SE","submitted_at":"2023-10-05T13:33:26Z","cross_cats_sorted":[],"title_canon_sha256":"8efcf10ed52b4d19facb87fa3de0cb9c209e2579466e05dfe69f9f2dca2fae4d","abstract_canon_sha256":"42e79401e8a78d4bc9c825c93cec45eb00c42db1f06b8eb7fc6fb30968788ef2"},"schema_version":"1.0"},"receipt":{"kind":"pith_receipt","key_id":"pith-v1-2026-05","algorithm":"ed25519","signed_at":"2026-07-05T07:11:36.710338Z","signature_b64":"NNQcdFPKUW650hUGVVBRapa/QGkkGrDbCSWy4g2v2GJQo+moGH51y2/PnoeH+8r8gXBTYut5rweeoVOgwCs1Bw==","signed_message":"canonical_sha256_bytes","builder_version":"pith-number-builder-2026-05-17-v1","receipt_version":"0.3","canonical_sha256":"a25191a59b060c05aaee24cc85f1d366fc6a17ddc374bb9a02c13e3fa16f9a6e","last_reissued_at":"2026-07-05T07:11:36.709613Z","signature_status":"signed_v1","first_computed_at":"2026-07-05T07:11:36.709613Z","public_key_fingerprint":"8d4b5ee74e4693bcd1df2446408b0d54"},"graph_snapshot":{"paper":{"title":"Large Language Models for Software Engineering: Survey and Open Problems","license":"http://creativecommons.org/licenses/by/4.0/","headline":"","cross_cats":[],"primary_cat":"cs.SE","authors_text":"Angela Fan, Beliz Gokkaya, Jie M. Zhang, Mark Harman, Mitya Lyubarskiy, Shin Yoo, Shubho Sengupta","submitted_at":"2023-10-05T13:33:26Z","abstract_excerpt":"This paper provides a survey of the emerging area of Large Language Models (LLMs) for Software Engineering (SE). It also sets out open research challenges for the application of LLMs to technical problems faced by software engineers. LLMs' emergent properties bring novelty and creativity with applications right across the spectrum of Software Engineering activities including coding, design, requirements, repair, refactoring, performance improvement, documentation and analytics. However, these very same emergent properties also pose significant technical challenges; we need techniques that can "},"claims":{"count":0,"items":[],"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"source":{"id":"2310.03533","kind":"arxiv","version":4},"verdict":{"id":null,"model_set":{},"created_at":null,"strongest_claim":"","one_line_summary":"","pipeline_version":null,"weakest_assumption":"","pith_extraction_headline":""},"integrity":{"clean":true,"summary":{"advisory":0,"critical":0,"by_detector":{},"informational":0},"endpoint":"/pith/2310.03533/integrity.json","findings":[],"available":true,"detectors_run":[],"snapshot_sha256":"c28c3603d3b5d939e8dc4c7e95fa8dfce3d595e45f758748cecf8e644a296938"},"references":{"count":0,"sample":[],"resolved_work":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57","internal_anchors":0},"formal_canon":{"evidence_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"author_claims":{"count":0,"strong_count":0,"snapshot_sha256":"258153158e38e3291e3d48162225fcdb2d5a3ed65a07baac614ab91432fd4f57"},"builder_version":"pith-number-builder-2026-05-17-v1"},"aliases":[{"alias_kind":"arxiv","alias_value":"2310.03533","created_at":"2026-07-05T07:11:36.709693+00:00"},{"alias_kind":"arxiv_version","alias_value":"2310.03533v4","created_at":"2026-07-05T07:11:36.709693+00:00"},{"alias_kind":"doi","alias_value":"10.48550/arxiv.2310.03533","created_at":"2026-07-05T07:11:36.709693+00:00"},{"alias_kind":"pith_short_12","alias_value":"UJIZDJM3AYGA","created_at":"2026-07-05T07:11:36.709693+00:00"},{"alias_kind":"pith_short_16","alias_value":"UJIZDJM3AYGALKXO","created_at":"2026-07-05T07:11:36.709693+00:00"},{"alias_kind":"pith_short_8","alias_value":"UJIZDJM3","created_at":"2026-07-05T07:11:36.709693+00:00"}],"events":[],"event_summary":{},"paper_claims":[],"inbound_citations":{"count":15,"internal_anchor_count":1,"sample":[{"citing_arxiv_id":"2607.05708","citing_title":"Akashic: A Low-Overhead LLM Inference Service with MemAttention","ref_index":14,"is_internal_anchor":true},{"citing_arxiv_id":"2606.17203","citing_title":"Trust-Aware Multi-Agent Traceability: Confidence-Calibrated Knowledge Graphs for Consistent Software Artifact Management","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2606.29124","citing_title":"CornerCase: Automated Extremal Testing of Protocol Implementations using LLMs","ref_index":27,"is_internal_anchor":false},{"citing_arxiv_id":"2305.12138","citing_title":"Exploring Code Analysis: Zero-Shot Insights on Syntax and Semantics with LLMs","ref_index":30,"is_internal_anchor":false},{"citing_arxiv_id":"2404.01535","citing_title":"Assessing, Exploiting, and Mitigating Syntactic Robustness Failures in LLM-Based Code Generation","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2411.09916","citing_title":"\"Should I Give Up Now?\" Investigating LLM Pitfalls in Software Engineering","ref_index":14,"is_internal_anchor":false},{"citing_arxiv_id":"2411.10656","citing_title":"Precision or Peril: A PoC of Python Code Quality from Quantized Large Language Models","ref_index":1,"is_internal_anchor":false},{"citing_arxiv_id":"2603.00989","citing_title":"Sustainable Code Generation Using Large Language Models: A Systematic Literature Review","ref_index":63,"is_internal_anchor":false},{"citing_arxiv_id":"2404.13501","citing_title":"A Survey on the Memory Mechanism of Large Language Model based Agents","ref_index":39,"is_internal_anchor":false},{"citing_arxiv_id":"2604.02548","citing_title":"From Theory to Practice: Code Generation Using LLMs for CAPEC and CWE Frameworks","ref_index":18,"is_internal_anchor":false},{"citing_arxiv_id":"2402.19173","citing_title":"StarCoder 2 and The Stack v2: The Next Generation","ref_index":198,"is_internal_anchor":false},{"citing_arxiv_id":"2604.00239","citing_title":"A Taxonomy of Programming Languages for Code Generation","ref_index":4,"is_internal_anchor":false},{"citing_arxiv_id":"2403.07974","citing_title":"LiveCodeBench: Holistic and Contamination Free Evaluation of Large Language Models for Code","ref_index":46,"is_internal_anchor":false},{"citing_arxiv_id":"2604.13725","citing_title":"On the Effectiveness of Context Compression for Repository-Level Tasks: An Empirical Investigation","ref_index":6,"is_internal_anchor":false},{"citing_arxiv_id":"2604.16790","citing_title":"Bias in the Loop: Auditing LLM-as-a-Judge for Software Engineering","ref_index":7,"is_internal_anchor":false}]},"formal_canon":{"evidence_count":0,"sample":[],"anchors":[]},"links":{"html":"https://pith.science/pith/UJIZDJM3AYGALKXOETGIL4OTM3","json":"https://pith.science/pith/UJIZDJM3AYGALKXOETGIL4OTM3.json","graph_json":"https://pith.science/api/pith-number/UJIZDJM3AYGALKXOETGIL4OTM3/graph.json","events_json":"https://pith.science/api/pith-number/UJIZDJM3AYGALKXOETGIL4OTM3/events.json","paper":"https://pith.science/paper/UJIZDJM3"},"agent_actions":{"view_html":"https://pith.science/pith/UJIZDJM3AYGALKXOETGIL4OTM3","download_json":"https://pith.science/pith/UJIZDJM3AYGALKXOETGIL4OTM3.json","view_paper":"https://pith.science/paper/UJIZDJM3","resolve_alias":"https://pith.science/api/pith-number/resolve?arxiv=2310.03533&json=true","fetch_graph":"https://pith.science/api/pith-number/UJIZDJM3AYGALKXOETGIL4OTM3/graph.json","fetch_events":"https://pith.science/api/pith-number/UJIZDJM3AYGALKXOETGIL4OTM3/events.json","actions":{"anchor_timestamp":"https://pith.science/pith/UJIZDJM3AYGALKXOETGIL4OTM3/action/timestamp_anchor","attest_storage":"https://pith.science/pith/UJIZDJM3AYGALKXOETGIL4OTM3/action/storage_attestation","attest_author":"https://pith.science/pith/UJIZDJM3AYGALKXOETGIL4OTM3/action/author_attestation","sign_citation":"https://pith.science/pith/UJIZDJM3AYGALKXOETGIL4OTM3/action/citation_signature","submit_replication":"https://pith.science/pith/UJIZDJM3AYGALKXOETGIL4OTM3/action/replication_record"}},"created_at":"2026-07-05T07:11:36.709693+00:00","updated_at":"2026-07-05T07:11:36.709693+00:00"}