{"total":18,"items":[{"citing_arxiv_id":"2606.30906","ref_index":16,"ref_count":1,"confidence":0.9,"is_internal_anchor":false,"paper_title":"Investigating Multi-Agent Deliberation in Law","primary_cat":"cs.AI","submitted_at":"2026-06-29T20:56:37+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":5.0,"formal_verification":"none","one_line_summary":"Multi-agent deliberation frameworks for legal reasoning with LLMs match baseline performance but yield distinct answers that cover cases single models miss.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.25442","ref_index":52,"ref_count":1,"confidence":0.9,"is_internal_anchor":false,"paper_title":"PolicyAlign: Direct Policy-Based Safety Alignment for Large Language Models","primary_cat":"cs.CL","submitted_at":"2026-06-24T06:10:33+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":5.0,"formal_verification":"none","one_line_summary":"PolicyAlign aligns LLMs to natural-language safety policies by synthesizing violating instructions and performing on-policy self-distillation with policy-sensitive filtering, improving safety without high-quality supervision data.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.23375","ref_index":33,"ref_count":2,"confidence":0.9,"is_internal_anchor":false,"paper_title":"Measuring & Mitigating Over-Alignment for LLMs in Multilingual Criminal Law Courts","primary_cat":"cs.CL","submitted_at":"2026-06-22T14:08:08+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"Creates TF-RefusalBench to quantify over-alignment in LLMs on criminal-law tasks across four languages and shows abliteration mitigates refusals with little performance loss.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.18021","ref_index":22,"ref_count":1,"confidence":0.9,"is_internal_anchor":false,"paper_title":"LegalHalluLens: Typed Hallucination Auditing and Calibrated Multi-Agent Debate for Trustworthy Legal AI","primary_cat":"cs.AI","submitted_at":"2026-06-16T15:02:37+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"LegalHalluLens provides typed hallucination profiles over CUAD, a Risk Direction Index, and a calibrated debate pipeline that reveals 38-40 pp category gaps hidden by aggregate 52% error rates and reduces fabricated detections by 45%.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.23716","ref_index":2,"ref_count":1,"confidence":0.9,"is_internal_anchor":false,"paper_title":"Legal Reasoning Is Not Lawyering: Rethinking Legal Benchmarks for Pro Se Access to Justice","primary_cat":"cs.CY","submitted_at":"2026-06-16T14:19:04+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":5.0,"formal_verification":"none","one_line_summary":"Legal AI benchmarks must evaluate robustness to pro se litigant inputs rather than expert-preprocessed ones to support access-to-justice claims.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.08932","ref_index":11,"ref_count":1,"confidence":0.9,"is_internal_anchor":false,"paper_title":"From Statute to Control Flow: Span-Grounded Deontic Trees for Defeasible Scope Parsing","primary_cat":"cs.CL","submitted_at":"2026-06-08T02:17:53+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"Introduces NormBench benchmark and Span-Grounded Deontic Trees (SG-DT) for defeasible scope parsing to reduce Silent Scope Omission in LLMs on statutes and policies.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.06941","ref_index":17,"ref_count":1,"confidence":0.9,"is_internal_anchor":false,"paper_title":"Quantum-Inspired Trace-Augmented Evidence Selection for Reasoning over Structured Hypothesis Spaces","primary_cat":"cs.AI","submitted_at":"2026-06-05T06:12:11+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":5.0,"formal_verification":"none","one_line_summary":"EP-HUBO treats CoT evidence selection as higher-order unconstrained binary optimization over per-hypothesis pools with quality weights to improve aggregation on legal benchmarks.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.03131","ref_index":14,"ref_count":1,"confidence":0.9,"is_internal_anchor":false,"paper_title":"HARVE: Hacking-Aware Reward-Head Vector Editing for Robust Reward Models","primary_cat":"cs.LG","submitted_at":"2026-06-02T04:18:08+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"HARVE removes the component of the reward-head vector aligned with a multi-directional hacking subspace from residual streams using a small set of contrastive examples, improving robustness on RewardHackBench across eight models without fine-tuning while preserving general capability.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2605.23497","ref_index":5,"ref_count":1,"confidence":0.9,"is_internal_anchor":false,"paper_title":"Asking For An Old Friend: Diagnosing and Mitigating Temporal Failure Modes in LLM-based Statutory Question Answering","primary_cat":"cs.CL","submitted_at":"2026-05-22T11:02:01+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":7.0,"formal_verification":"none","one_line_summary":"LLMs show severe staleness after training cutoffs and recency bias on historical German statutes; RAG with version filtering mitigates both better than web search.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2605.21076","ref_index":16,"ref_count":1,"confidence":0.9,"is_internal_anchor":false,"paper_title":"GradeLegal: Automated Grading for German Legal Cases","primary_cat":"cs.CL","submitted_at":"2026-05-20T12:09:49+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":5.0,"formal_verification":"none","one_line_summary":"Reasoning-oriented LLMs reach up to 0.91 quadratic weighted kappa agreement with experts on public law cases when given sample solutions and grading rubrics, but only 0.60 on criminal law cases.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2605.10186","ref_index":8,"ref_count":1,"confidence":0.9,"is_internal_anchor":false,"paper_title":"LegalCiteBench: Evaluating Citation Reliability in Legal Language Models","primary_cat":"cs.CL","submitted_at":"2026-05-11T08:37:54+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"LegalCiteBench reveals that current LLMs achieve under 7% accuracy on closed-book legal citation retrieval and completion tasks, with misleading answer rates above 94% for nearly all tested models.","context_count":1,"top_context_role":"background","top_context_polarity":"background","context_text":"Dong Nguyen, Thang V . Q. Le, Nguyen P. Nguyen, et al. LawLLM: Law large language model for the US legal system. InProceedings of the 33rd ACM International Conference on Information and Knowledge Management, 2024. Cynthia A Norton and Nancy B Rapoport. Doubling down on dumb: Lessons from mata v. avianca inc.American Bankruptcy Institute Journal, 42(8):24-61, 2023. Yuzhen Shi, Huanghai Liu, Yiran Hu, Gaojie Song, Xinran Xu, Yubo Ma, Tianyi Tang, Li Zhang, Qingjing Chen, Di Feng, et al. Plawbench: A rubric-based benchmark for evaluating llms in real-world legal practice.arXiv preprint arXiv:2601.16669, 2026. An Yang, Anfeng Li, Baosong Yang, Beichen Zhang, Binyuan Hui, Bo Zheng, Bowen Yu, Chang Gao, Chengen Huang, Chenxu Lv, et al."},{"citing_arxiv_id":"2605.08437","ref_index":4,"ref_count":1,"confidence":0.9,"is_internal_anchor":false,"paper_title":"Magis-Bench: Evaluating LLMs on Magistrate-Level Legal Tasks","primary_cat":"cs.CL","submitted_at":"2026-05-08T20:00:14+00:00","verdict":"ACCEPT","verdict_confidence":"LOW","novelty_score":7.0,"formal_verification":"none","one_line_summary":"Magis-Bench is a new benchmark of 74 magistrate-level legal writing tasks from Brazilian exams where the strongest LLMs reach only 6.97/10, showing judicial reasoning remains difficult for current models.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2604.23730","ref_index":6,"ref_count":1,"confidence":0.9,"is_internal_anchor":false,"paper_title":"Expert Evaluation of LLM's Open-Ended Legal Reasoning on the Japanese Bar Exam Writing Task","primary_cat":"cs.AI","submitted_at":"2026-04-26T14:15:43+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":7.0,"formal_verification":"none","one_line_summary":"Expert evaluation of LLMs on Japanese bar exam writing tasks shows clear limitations in open-ended legal reasoning and frequent hallucinations unsupported by law or precedent.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2604.20726","ref_index":4,"ref_count":1,"confidence":0.9,"is_internal_anchor":false,"paper_title":"Exploiting LLM-as-a-Judge Disposition on Free Text Legal QA via Prompt Optimization","primary_cat":"cs.CL","submitted_at":"2026-04-22T16:12:36+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":5.0,"formal_verification":"none","one_line_summary":"Automatic prompt optimization using lenient LLM judges improves performance and transferability in legal QA evaluations compared to human design or strict judges.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2604.13583","ref_index":1,"ref_count":1,"confidence":0.9,"is_internal_anchor":false,"paper_title":"BenGER Platform: A Collaborative Web Platform for End-to-End Benchmarking of German Legal Tasks","primary_cat":"cs.CL","submitted_at":"2026-04-15T07:43:01+00:00","verdict":"CONDITIONAL","verdict_confidence":"HIGH","novelty_score":4.0,"formal_verification":"none","one_line_summary":"BenGER integrates task creation, annotation, configurable LLM runs, and lexical/semantic/factual/judge metrics into a multi-tenant web platform for German legal benchmarking.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2605.16280","ref_index":14,"ref_count":1,"confidence":0.9,"is_internal_anchor":false,"paper_title":"Beyond Imperfect Alternatives with Rulemapping: A Neuro-Symbolic Case Study on Online Hate Speech","primary_cat":"cs.CY","submitted_at":"2026-04-10T17:24:34+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":5.0,"formal_verification":"none","one_line_summary":"Rulemapping uses expert symbolic scaffolds to constrain LLMs, raising precision on §130(1) German hate-speech classification from 0.34-0.49 to 0.80-0.86 while preserving recall of 0.82-0.89.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2604.09069","ref_index":4,"ref_count":1,"confidence":0.9,"is_internal_anchor":false,"paper_title":"NyayaMind- A Framework for Transparent Legal Reasoning and Judgment Prediction in the Indian Legal System","primary_cat":"cs.CL","submitted_at":"2026-04-10T07:51:42+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":5.0,"formal_verification":"none","one_line_summary":"NyayaMind combines RAG retrieval with domain-specific LLMs to generate transparent, structured legal reasoning and judgment predictions for Indian court cases.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2602.09514","ref_index":9,"ref_count":1,"confidence":0.9,"is_internal_anchor":false,"paper_title":"EcoGym: Evaluating LLMs for Long-Horizon Plan-and-Execute in Interactive Economies","primary_cat":"cs.CL","submitted_at":"2026-02-10T08:12:23+00:00","verdict":"CONDITIONAL","verdict_confidence":"MODERATE","novelty_score":6.0,"formal_verification":"none","one_line_summary":"EcoGym is a new open benchmark with three economic environments that reveals no leading LLM dominates at sustained plan-and-execute decision making across scenarios.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null}],"limit":50,"offset":0}