{"total":26,"items":[{"citing_arxiv_id":"2607.07748","ref_index":20,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Selective Left-Shift: Turning Test-Time Compute and Difficulty-based Curation into Training Data for Low-Resource Code Generation","primary_cat":"cs.LG","submitted_at":"2026-07-08T09:18:56+00:00","verdict":"CONDITIONAL","verdict_confidence":"MODERATE","novelty_score":6.0,"formal_verification":"none","one_line_summary":"Left-shifting iterative compiler/test refinement into verified SFT data, then GRPO on difficulty-curated IO rewards, lifts Qwen3-8B Julia pass@1 past prior SOTA at 1/3 data and 1/6 cost, and bootstraps Ballerina.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2607.06009","ref_index":23,"ref_count":1,"confidence":0.98,"is_internal_anchor":true,"paper_title":"Multi-Channel Spread-Spectrum Code Watermarking","primary_cat":"cs.CR","submitted_at":"2026-07-07T08:50:19+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":7.0,"formal_verification":"none","one_line_summary":"A training-free post-hoc code watermark embeds 24-bit identifiers via multi-channel spread-spectrum encoding over naming conventions and semantic pattern pairs, with majority voting and Reed-Solomon recovery.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.17683","ref_index":8,"ref_count":1,"confidence":0.9,"is_internal_anchor":false,"paper_title":"Bridging Functional Correctness and Runtime Efficiency Gaps in LLM-Based Code Translation","primary_cat":"cs.CL","submitted_at":"2026-06-16T08:49:56+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":7.0,"formal_verification":"none","one_line_summary":"SwiftTrans improves both functional correctness and runtime efficiency of LLM code translations via multi-perspective exploration with hierarchical guidance and difference-aware selection with ordinal guidance on extended benchmarks including new SwiftBench.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.11755","ref_index":32,"ref_count":1,"confidence":0.9,"is_internal_anchor":false,"paper_title":"Acoda: Adversarial Code Obfuscation for Defending against LLM-based Analysis","primary_cat":"cs.SE","submitted_at":"2026-06-10T07:29:31+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"Acoda uses a genetic algorithm to optimize eight obfuscation methods that reduce LLM code analysis success rates to as low as 30% while preserving original semantics.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.08840","ref_index":18,"ref_count":1,"confidence":0.9,"is_internal_anchor":false,"paper_title":"Beyond Pass Rate: A Multilingual, Execution-Grounded Evaluation of Open Code LLMs","primary_cat":"cs.AI","submitted_at":"2026-06-07T21:10:30+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":7.0,"formal_verification":"none","one_line_summary":"Multilingual execution-grounded benchmark finds top open code LLM at 23.64% correctness versus 57.2% human baseline, with compile errors dominating 63% of failures.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.06821","ref_index":34,"ref_count":1,"confidence":0.9,"is_internal_anchor":false,"paper_title":"Chiseling Out Efficiency: Structured Skeleton Supervision for Efficient Code Generation","primary_cat":"cs.SE","submitted_at":"2026-06-05T01:49:20+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"EffiSkel improves LLM-generated code efficiency by supervising on extracted structural efficiency skeletons via multi-task learning of code generation and skeleton prediction.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.01723","ref_index":24,"ref_count":1,"confidence":0.9,"is_internal_anchor":false,"paper_title":"Shortcut to Nowhere: Demystifying Deep Spurious Regression","primary_cat":"cs.LG","submitted_at":"2026-06-01T05:40:11+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"Defines DSR and introduces similarity-based calibration of label and feature distributions to mitigate continuous spurious correlations in regression.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2605.28409","ref_index":10,"ref_count":1,"confidence":0.9,"is_internal_anchor":false,"paper_title":"Efficient Post-training of LLMs for Code Generation With Offline Reinforcement Learning","primary_cat":"cs.AI","submitted_at":"2026-05-27T12:43:14+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":4.0,"formal_verification":"none","one_line_summary":"Offline RL post-training boosts code generation performance in LLMs, with larger gains for small models and hard problems, using pre-collected datasets.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2605.13896","ref_index":12,"ref_count":1,"confidence":0.9,"is_internal_anchor":false,"paper_title":"Neural Code Translation of Legacy Code: APL to C#","primary_cat":"cs.SE","submitted_at":"2026-05-12T12:11:33+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":5.0,"formal_verification":"none","one_line_summary":"Guided LLM strategies with custom datasets and execution-based verification enable functional APL-to-C# translation across a range of program complexities.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2605.09421","ref_index":1,"ref_count":2,"confidence":0.9,"is_internal_anchor":false,"paper_title":"MACAA: Belief-Revision Multi-Agent Reasoning for Code Authorship Verification","primary_cat":"cs.SE","submitted_at":"2026-05-10T08:47:05+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":7.0,"formal_verification":"none","one_line_summary":"MACAA is a belief-revision multi-agent framework for training-free code authorship verification that reports 89.15% F1 on same-language benchmarks and 80% on cross-language pairs while outperforming baselines.","context_count":1,"top_context_role":"background","top_context_polarity":"background","context_text":"cause conflict without unnecessary changes to the overall decision process. 3.2 Problem Formulation Given two code samplesx1 and x2, potentially writ- ten in distinct programming languages, MACAA predicts whether they originate from the same au- thor. The final decision includes: (1) a binary la- bel is_same_author∈ {true,false} ; (2) a con- fidence score c∈[0,1] ; (3) an evidence chain E={e 1, . . . , ek}documenting the belief-revision trajectory; (4) verdict reasoning that summarizes the decisive evidence; and (5) dissenting opinions that record unresolved conflicts, weak signals, or reliability concerns for forensic auditability. 3.3 Coordinator Agent The Coordinator Agent implements the state- machine flow shown by the red arrows in Fig-"},{"citing_arxiv_id":"2605.02860","ref_index":37,"ref_count":1,"confidence":0.9,"is_internal_anchor":false,"paper_title":"Standing on the Shoulders of Giants: Stabilized Knowledge Distillation for Cross--Language Code Clone Detection","primary_cat":"cs.AI","submitted_at":"2026-05-04T17:37:16+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":5.0,"formal_verification":"none","one_line_summary":"Reasoning-oriented knowledge distillation from DeepSeek-R1 plus response stabilization improves reliability and often performance of compact models for cross-language code clone detection on pairs like Python-Java and Rust-Java.","context_count":1,"top_context_role":"background","top_context_polarity":"background","context_text":"has been widely studied in NLP and computer vision [22, 36, 46], and it is increasingly relevant to software engineering because code intelligence tasks often require efficient inference over large repositories. Prior SE works such as Compressor and Avatar use KD to compress code models and optimize student configurations for code intelligence tasks [37, 38]. Recent work has begun to study KD more systematically for code understanding. Wang et al. [45] evaluate logit-based and feature-based KD methods across multiple student architectures and teacher models on defect detection, clone detection, and exception classification. They show that KD generally improves student models over standard fine-tuning, that code-specific teachers are"},{"citing_arxiv_id":"2605.02195","ref_index":25,"ref_count":2,"confidence":0.9,"is_internal_anchor":false,"paper_title":"Beyond Translation Accuracy: Addressing False Failures in LLM-Based Code Translation","primary_cat":"cs.SE","submitted_at":"2026-05-04T03:49:58+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":5.0,"formal_verification":"none","one_line_summary":"A large-scale study finds that many LLM code translation failures are false negatives due to improper evaluation configurations rather than incorrect translations.","context_count":1,"top_context_role":"background","top_context_polarity":"background","context_text":"Vladimir Zolotov, Julian Dolby, Jie Chen, Mihir Choudhury, Lindsey Decker, et al. 2021. Codenet: A large-scale ai for code dataset for learning a diversity of coding tasks. arXiv:2105.12655 [cs.SE] https://arxiv.org/abs/2105.12655 [24] Fazle Rabbi, Zishuo Ding, and Jinqiu Yang. 2025. A Multi-Language Perspective on the Robustness of LLM Code Generation. arXiv:2504.19108 [cs.SE] https: //arxiv.org/abs/2504.19108 [25] Fazle Rabbi, Lin Ling, Song Wang, and Jinqiu Yang. 2026. Social Bias in LLM- Generated Code: Benchmark and Mitigation. arXiv:2605.00382 [cs.SE] https: //arxiv.org/abs/2605.00382 [26] Fazle Rabbi, Soumit Kanti Saha, Tri Minh Triet Pham, Song Wang, and Jinqiu Yang. 2025. BabelCoder: Agentic Code Translation with Specification Alignment. arXiv:2512.06902 [cs."},{"citing_arxiv_id":"2604.25599","ref_index":25,"ref_count":1,"confidence":0.9,"is_internal_anchor":false,"paper_title":"PLMGH: What Matters in PLM-GNN Hybrids for Code Classification and Vulnerability Detection","primary_cat":"cs.SE","submitted_at":"2026-04-28T13:05:36+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":5.0,"formal_verification":"none","one_line_summary":"Controlled experiments show PLM-GNN hybrids improve code tasks over GNN-only baselines, with PLM source having larger impact than GNN backbone.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2604.25960","ref_index":34,"ref_count":1,"confidence":0.9,"is_internal_anchor":false,"paper_title":"Large Language Models for Multilingual Code Intelligence: A Survey","primary_cat":"cs.SE","submitted_at":"2026-04-27T20:20:26+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":4.0,"formal_verification":"none","one_line_summary":"A survey of methods, benchmarks, and open challenges for large language models in multilingual code generation and translation.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2604.18027","ref_index":51,"ref_count":1,"confidence":0.9,"is_internal_anchor":false,"paper_title":"CodePivot: Bootstrapping Multilingual Transpilation in LLMs via Reinforcement Learning without Parallel Corpora","primary_cat":"cs.SE","submitted_at":"2026-04-20T09:52:50+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"CodePivot uses Python as a pivot language plus an Aggressive-Partial-Functional RL reward to train a 7B model that outperforms much larger LLMs on multilingual code transpilation without parallel corpora.","context_count":1,"top_context_role":"background","top_context_polarity":"background","context_text":"that require hand-crafted rules for each translation setting, which limit scalability and make it difficult to generalize across languages and complex programming patterns. More recently, learning-based and LLM-based approaches have become an active direction [ 78, 48, 5, 47, 52, 9]. Prior work strengthens the core code translation capability of LLMs by leveraging large-scale data and improved training algorithms [ 51, 89, 53, 50, 66, 8], while other studies rely on prompting and pipeline orchestration to handle challenging cases [38, 81]. Although these methods often generalize well, they typically provide weaker guarantees of correctness. Some approaches investigate the synergy between rule-based methods and LLMs to balance correctness with generalization [23, 4, 79]."},{"citing_arxiv_id":"2604.16058","ref_index":18,"ref_count":1,"confidence":0.9,"is_internal_anchor":false,"paper_title":"LLMSniffer: Detecting LLM-Generated Code via GraphCodeBERT and Supervised Contrastive Learning","primary_cat":"cs.SE","submitted_at":"2026-04-17T13:32:25+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":4.0,"formal_verification":"none","one_line_summary":"LLMSniffer improves detection of LLM-generated code on GPTSniffer and Whodunit benchmarks by fine-tuning GraphCodeBERT via two-stage supervised contrastive learning plus preprocessing and MLP classification.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2604.08083","ref_index":57,"ref_count":1,"confidence":0.9,"is_internal_anchor":false,"paper_title":"Can LLMs Deobfuscate Binary Code? A Systematic Analysis of Large Language Models into Pseudocode Deobfuscation","primary_cat":"cs.SE","submitted_at":"2026-04-09T10:56:06+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":7.0,"formal_verification":"none","one_line_summary":"LLM deobfuscation of binaries to pseudocode depends more on reasoning ability and task-specific fine-tuning than on model size, with reasoning models showing robustness across ISAs and obfuscation levels on the new BinDeObfBench.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2603.16011","ref_index":1,"ref_count":1,"confidence":0.9,"is_internal_anchor":false,"paper_title":"FormulaCode: Evaluating Agentic Optimization on Large Codebases","primary_cat":"cs.SE","submitted_at":"2026-03-16T23:40:19+00:00","verdict":"CONDITIONAL","verdict_confidence":"MODERATE","novelty_score":7.0,"formal_verification":"none","one_line_summary":"FormulaCode is a 957-task benchmark showing that frontier LLM coding agents speed up real scientific Python codebases but still fall short of the human expert patches on multi-workload performance.","context_count":1,"top_context_role":"background","top_context_polarity":"unclear","context_text":"single parameter is needed, leading to severe performance penalties for large parameter tables. 14Comments: 15 Currently in draft because there's no tests - I'm just putting it up so Sam and Ian from #14471 can test it out for their use case. For the explicit example in that issue, a complete comparison on my machine: 16 <details><summary>Out of date timings</summary> 17 In [1]: from qiskit.circuit import Parameter, ParameterExpression 18 N: int = 100_000 19 parameter_values = {Parameter(f\"th_{i}\"): 1 for i in range(N)} 20 parameter_values[param := Parameter(\"my_param\")] = 1 21 . . .<TRUNCATED> 22 I think it's fine without having the same behavior. For clarity it might be helpful to add a blurb to the bind_all docstring to say that \"unlike bind, NaN and inf are in the range of expected outputs for this method\"."},{"citing_arxiv_id":"2604.02352","ref_index":22,"ref_count":1,"confidence":0.9,"is_internal_anchor":false,"paper_title":"An Initial Exploration of Contrastive Prompt Tuning to Generate Energy-Efficient Code","primary_cat":"cs.LG","submitted_at":"2026-03-03T12:36:15+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":5.0,"formal_verification":"none","one_line_summary":"Contrastive Prompt Tuning raises code accuracy on two of three tested models but produces inconsistent energy-efficiency gains that depend on model, language, and task.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2507.21954","ref_index":55,"ref_count":1,"confidence":0.9,"is_internal_anchor":false,"paper_title":"Fine-Tuning Code Language Models to Detect Cross-Language Bugs","primary_cat":"cs.SE","submitted_at":"2025-07-29T16:06:08+00:00","verdict":"CONDITIONAL","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"Fine-tuning 13 CodeLMs on a constructed CLB dataset with nine interaction types improves detection, with UniXcoder-base reaching F1 0.7407 and small models outperforming large ones.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2505.10708","ref_index":32,"ref_count":1,"confidence":0.9,"is_internal_anchor":false,"paper_title":"SafeTrans: LLM-assisted Transpilation from C to Rust","primary_cat":"cs.CR","submitted_at":"2025-05-15T21:05:33+00:00","verdict":"ACCEPT","verdict_confidence":"MODERATE","novelty_score":6.0,"formal_verification":"none","one_line_summary":"SafeTrans achieves up to 80% successful C-to-Rust translations via LLM iterative repair on 2653 programs and two real projects, with some C vulnerabilities carrying over to the Rust output.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2412.14399","ref_index":34,"ref_count":1,"confidence":0.9,"is_internal_anchor":false,"paper_title":"NESA: Relational Neuro-Symbolic Static Program Analysis","primary_cat":"cs.PL","submitted_at":"2024-12-18T23:14:59+00:00","verdict":"CONDITIONAL","verdict_confidence":"LOW","novelty_score":7.0,"formal_verification":"none","one_line_summary":"NESA presents a neuro-symbolic framework that decomposes static analyses into policy-defined sub-problems solved by parsers and LLMs to enable compilation-free customizable analysis with reduced hallucinations.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2412.04590","ref_index":18,"ref_count":1,"confidence":0.9,"is_internal_anchor":false,"paper_title":"Specification-Driven Code Translation Powered by Large Language Models: How Far Are We?","primary_cat":"cs.SE","submitted_at":"2024-12-05T20:10:21+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":4.0,"formal_verification":"none","one_line_summary":"NL specifications alone do not improve LLM code translation performance, but combining them with source code yields gains in select language pairs with no overall consistent benefit.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2409.19894","ref_index":34,"ref_count":1,"confidence":0.9,"is_internal_anchor":false,"paper_title":"TransAgent: Enhancing LLM-Based Code Translation via Fine-Grained Execution Alignment","primary_cat":"cs.SE","submitted_at":"2024-09-30T02:53:03+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":5.0,"formal_verification":"none","one_line_summary":"TransAgent improves LLM code translation by up to 33.3% via multi-agent fine-grained execution alignment on a new benchmark of recent tasks.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2303.17651","ref_index":34,"ref_count":1,"confidence":0.9,"is_internal_anchor":false,"paper_title":"Self-Refine: Iterative Refinement with Self-Feedback","primary_cat":"cs.CL","submitted_at":"2023-03-30T18:30:01+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":5.0,"formal_verification":"none","one_line_summary":"Self-Refine boosts LLM outputs by ~20% on average across seven tasks by having the same model iteratively generate, critique, and refine its own responses.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2112.00114","ref_index":13,"ref_count":1,"confidence":0.9,"is_internal_anchor":false,"paper_title":"Show Your Work: Scratchpads for Intermediate Computation with Language Models","primary_cat":"cs.LG","submitted_at":"2021-11-30T21:32:46+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":8.0,"formal_verification":"none","one_line_summary":"Training language models to generate intermediate computation steps on a scratchpad enables them to perform multi-step tasks such as long addition and arbitrary program execution that they otherwise fail at.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null}],"limit":50,"offset":0}