{"total":19,"items":[{"citing_arxiv_id":"2607.07881","ref_index":29,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Functional and Secure Code Generation with Task Vectors","primary_cat":"cs.SE","submitted_at":"2026-07-08T19:32:22+00:00","verdict":"ACCEPT","verdict_confidence":"HIGH","novelty_score":6.0,"formal_verification":"none","one_line_summary":"LPO-derived Secure-Anchored task vectors raise simultaneous functional-and-secure code rates by 2.1–36 pp on six coding LLMs with near-zero inference overhead.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2607.02186","ref_index":23,"ref_count":1,"confidence":0.9,"is_internal_anchor":false,"paper_title":"UA-ChatDev: Uncertainty-Aware Multi-Agent Collaboration for Reliable Software Development","primary_cat":"cs.AI","submitted_at":"2026-07-02T13:56:57+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":4.0,"formal_verification":"none","one_line_summary":"UA-ChatDev integrates token-level uncertainty estimation and phase-aware verification into multi-agent software development and reports better benchmark scores than prior frameworks.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.29815","ref_index":19,"ref_count":1,"confidence":0.9,"is_internal_anchor":false,"paper_title":"SrDetection: A Self-Referential Framework for Data Leakage Detection in Code Large Language Models","primary_cat":"cs.CL","submitted_at":"2026-06-29T05:48:42+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"SrDetection detects data leakage in Code LLMs via contrast between original benchmark samples and their semantic variants, reporting F1 gains of 21.52 (gray-box) and 14.46 (black-box) over baselines in a controlled testbed.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.24901","ref_index":96,"ref_count":1,"confidence":0.9,"is_internal_anchor":false,"paper_title":"LLM Evolution as an Industry-Scale Ecosystem: A Lifecycle Perspective on Continual Learning","primary_cat":"cs.LG","submitted_at":"2026-06-12T13:44:48+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":5.0,"formal_verification":"none","one_line_summary":"The paper reformulates industrial continual learning for LLMs as a closed-loop ecosystem problem, identifies three core challenges, and organizes solutions around five lifecycle design principles.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.11755","ref_index":34,"ref_count":1,"confidence":0.9,"is_internal_anchor":false,"paper_title":"Acoda: Adversarial Code Obfuscation for Defending against LLM-based Analysis","primary_cat":"cs.SE","submitted_at":"2026-06-10T07:29:31+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"Acoda uses a genetic algorithm to optimize eight obfuscation methods that reduce LLM code analysis success rates to as low as 30% while preserving original semantics.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.09145","ref_index":52,"ref_count":1,"confidence":0.9,"is_internal_anchor":false,"paper_title":"PrivCode++: Latent-Conditioned Differentially Private Code Generation for Comprehensive Guarantees","primary_cat":"cs.CR","submitted_at":"2026-06-08T07:42:44+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":7.0,"formal_verification":"none","one_line_summary":"PrivCode++ introduces the first DP code generation method protecting both prompts and code via latent-conditioned two-stage training, claiming higher utility and stronger privacy than prior baselines.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.07999","ref_index":75,"ref_count":1,"confidence":0.9,"is_internal_anchor":false,"paper_title":"Efficient Skill Grounding via Code Refactoring with Small Language Models","primary_cat":"cs.AI","submitted_at":"2026-06-06T06:33:51+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"RECENT decouples skill semantics from embodiment-specific bindings via code refactoring to let small language models achieve skill grounding performance matching large language model baselines.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.03128","ref_index":32,"ref_count":1,"confidence":0.9,"is_internal_anchor":false,"paper_title":"Decoupled Smart Contract Audits: Lightweight LLM Framework via Distillation and Aggregation","primary_cat":"cs.CR","submitted_at":"2026-06-02T04:13:43+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":5.0,"formal_verification":"none","one_line_summary":"A decoupled four-stage LLM pipeline with rsLoRA, distillation, and CoVe aggregation outperforms larger models on smart contract vulnerability detection and explanation using only 0.6B-4B parameter models.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2605.25296","ref_index":9,"ref_count":1,"confidence":0.9,"is_internal_anchor":false,"paper_title":"Subjective Code Preferences in Experts and Large Language Models","primary_cat":"cs.HC","submitted_at":"2026-05-24T23:20:12+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"LLMs frequently reverse their stated coding preferences when shown actual code instead of descriptions, show positional bias, and produce more polarized ratings than human experts on complexity, commenting, modularity, and readability.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2605.04894","ref_index":9,"ref_count":1,"confidence":0.9,"is_internal_anchor":false,"paper_title":"SynConfRoute: Syntax-Aware Routing for Efficient Code Completion with Small CodeLLMs","primary_cat":"cs.SE","submitted_at":"2026-05-06T13:25:34+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"SynConfRoute routes code completions using syntax validation and token confidence, improving pass@1 by up to 31% on hard tasks and reducing accelerator usage by 58% versus always using the largest model.","context_count":1,"top_context_role":"background","top_context_polarity":"background","context_text":"signal, opening a new class of signals beyond token confidence. 7 Related Work Our work builds on five areas: code language models, code comple- tion benchmarks, model routing, few-shot retrieval, and quantiza- tion. Code Language Models.Recent CodeLLMs including DeepSeek- Coder [18, 56], StarCoder2 [27], OpenCoder [20], Granite-Code [29], Yi-Coder [1], CodeGemma [9], CodeGeeX4 [55], Magicoder [50], and Qwen2.5-Coder [21] have advanced code generation through FIM training objectives [3], auxiliary objectives [11], and curriculum learning [42]. These models are deployed in industrial IDE systems with strict latency requirements: Meta's CodeCompose [31] fine- tunes CodeLlama for multi-line suggestions, JetBrains ships light-"},{"citing_arxiv_id":"2604.21365","ref_index":7,"ref_count":1,"confidence":0.9,"is_internal_anchor":false,"paper_title":"mcdok at SemEval-2026 Task 13: Finetuning LLMs for Detection of Machine-Generated Code","primary_cat":"cs.LG","submitted_at":"2026-04-23T07:29:06+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":2.0,"formal_verification":"none","one_line_summary":"Fine-tuning LLMs by adapting the mdok approach produces competitive results on binary detection, source attribution, and hybrid/adversarial code identification in SemEval-2026 Task 13.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2604.19826","ref_index":8,"ref_count":1,"confidence":0.9,"is_internal_anchor":false,"paper_title":"Co-Located Tests, Better AI Code: How Test Syntax Structure Affects Foundation Model Code Generation","primary_cat":"cs.SE","submitted_at":"2026-04-20T14:47:46+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"Co-locating tests with implementation code yields substantially higher preservation and correctness in foundation-model-generated programs than separated test syntax.","context_count":1,"top_context_role":"background","top_context_polarity":"background","context_text":"parameters) across model layers. For transformers, we extract post- softmax attention weights directly. For RWKV-6 (a gated-linear RNN with no attention matrices), we compute effective attention from the Weighted Key-Value (WKV) recurrence. Models.7 open-source models spanning diverse architectures: Qwen2.5-Coder-7B and -3B [12], StarCoder2-3B [19], CodeGemma- 7B [8], Code-LLaMA-7B [28], Phi-3-mini-4k-instruct [1] (6 trans- formers [34]), and RWKV-6-Finch-1B6 [25] (a gated-linear RNN [24]). MI requires access to internal representations that proprietary mod- els do not expose. The 7 models were selected for architectural diversity (including a non-transformer paradigm), code compe- tence, and feasibility on consumer hardware (16GB video RAM"},{"citing_arxiv_id":"2509.03117","ref_index":50,"ref_count":1,"confidence":0.9,"is_internal_anchor":false,"paper_title":"PromptCOS: Towards Content-only System Prompt Copyright Auditing for LLMs","primary_cat":"cs.CR","submitted_at":"2025-09-03T08:19:40+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":7.0,"formal_verification":"none","one_line_summary":"PromptCOS is a content-only watermarking method for LLM system prompts that embeds detectable cyclic signals via auxiliary tokens while preserving fidelity and resisting removal attacks.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2509.01082","ref_index":46,"ref_count":1,"confidence":0.9,"is_internal_anchor":false,"paper_title":"RefineStat: Efficient Exploration for Probabilistic Program Synthesis","primary_cat":"cs.LG","submitted_at":"2025-09-01T03:13:36+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"RefineStat improves small language model performance on probabilistic program synthesis by adding semantic constraint enforcement and diagnostic-aware refinement, producing syntactically and statistically reliable code that often matches larger models.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2507.06261","ref_index":15,"ref_count":1,"confidence":0.9,"is_internal_anchor":false,"paper_title":"Gemini 2.5: Pushing the Frontier with Advanced Reasoning, Multimodality, Long Context, and Next Generation Agentic Capabilities","primary_cat":"cs.CL","submitted_at":"2025-07-07T17:36:04+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":4.0,"formal_verification":"none","one_line_summary":"Gemini 2.5 Pro and Flash models are presented as achieving frontier performance in reasoning, coding, and long-context multimodal tasks while spanning a cost-capability Pareto curve.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2505.10443","ref_index":2,"ref_count":1,"confidence":0.9,"is_internal_anchor":false,"paper_title":"Are Large Language Models Robust in Understanding Code Against Semantics-Preserving Mutations?","primary_cat":"cs.SE","submitted_at":"2025-05-15T16:04:25+00:00","verdict":"CONDITIONAL","verdict_confidence":"MODERATE","novelty_score":6.0,"formal_verification":"none","one_line_summary":"LLMs achieve strong initial accuracy on code output prediction but frequently alter their answers under semantics-preserving mutations, with drops up to 70% and flawed reasoning detected in 10-50% of correct cases via human review.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2502.06556","ref_index":25,"ref_count":1,"confidence":0.9,"is_internal_anchor":false,"paper_title":"MultiFileTest: A Multi-File-Level LLM Unit Test Generation Benchmark and Impact of Error Fixing Mechanisms","primary_cat":"cs.SE","submitted_at":"2025-02-10T15:24:30+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"Frontier LLMs achieve only moderate performance on multi-file unit test generation, with basic executability and cascade errors common, but manual and self-error-fixing mechanisms yield measurable gains.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2410.22240","ref_index":64,"ref_count":1,"confidence":0.9,"is_internal_anchor":false,"paper_title":"Are Decoder-Only Large Language Models the Silver Bullet for Code Search?","primary_cat":"cs.SE","submitted_at":"2024-10-29T17:05:25+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":4.0,"formal_verification":"none","one_line_summary":"Fine-tuned decoder-only LLMs achieve up to 40.4% higher MAP than UniXcoder on CoSQA+ for code search, with non-monotonic size scaling and data composition sensitivity.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2409.12917","ref_index":32,"ref_count":1,"confidence":0.9,"is_internal_anchor":false,"paper_title":"Training Language Models to Self-Correct via Reinforcement Learning","primary_cat":"cs.LG","submitted_at":"2024-09-19T17:16:21+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"SCoRe uses multi-turn online RL with regularization on self-generated traces to improve LLM self-correction, achieving 15.6% and 9.1% gains on MATH and HumanEval for Gemini models.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null}],"limit":50,"offset":0}