{"total":15,"items":[{"citing_arxiv_id":"2607.05992","ref_index":49,"ref_count":1,"confidence":0.88,"is_internal_anchor":false,"paper_title":"PluraMath: Extending Mathematical Reasoning Evaluation Beyond High-Resource Languages","primary_cat":"cs.CL","submitted_at":"2026-07-07T08:25:29+00:00","verdict":"CONDITIONAL","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"PluraMath extends PolyMath with human-validated math problems in 18 mid-to-extreme low-resource languages and benchmarks 27 reasoning LLMs, finding a persistent high- vs low-resource performance gap.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.27019","ref_index":5,"ref_count":1,"confidence":0.88,"is_internal_anchor":false,"paper_title":"MinGram: A Minimalist Unigram Tokenizer with High Compression and Competitive Morphological Alignment","primary_cat":"cs.CL","submitted_at":"2026-06-25T13:31:02+00:00","verdict":null,"verdict_confidence":null,"novelty_score":null,"formal_verification":null,"one_line_summary":null,"context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.20993","ref_index":46,"ref_count":1,"confidence":0.88,"is_internal_anchor":false,"paper_title":"Phonemes to the Rescue: Multilingual Tokenization Based on International Phonetic Alphabet","primary_cat":"cs.CL","submitted_at":"2026-06-18T23:50:54+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"IPA-based subword tokenizers trained across 24 languages improve tokenization quality and generalization to unseen languages compared to standard text tokenizers, especially for non-Latin scripts.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.18389","ref_index":86,"ref_count":1,"confidence":0.88,"is_internal_anchor":false,"paper_title":"Want Better Synthetic Data? Steer It: Activation Steering for Low-Resource Language Generation","primary_cat":"cs.CL","submitted_at":"2026-06-16T18:34:21+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"Activation steering on early layers improves diversity of synthetic data for low-resource languages and often boosts downstream classifier performance compared to non-steered prompting.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.09767","ref_index":3,"ref_count":1,"confidence":0.88,"is_internal_anchor":false,"paper_title":"Data Synthesis and Parameter-Efficient Fine-Tuning for Low-Resource NMT: A Case Study on Q'eqchi' Mayan","primary_cat":"cs.CL","submitted_at":"2026-06-08T17:29:08+00:00","verdict":"UNVERDICTED","verdict_confidence":"UNKNOWN","novelty_score":4.0,"formal_verification":"none","one_line_summary":"Synthetic dictionary data with LoRA fine-tuning teaches structure to Q'eqchi' NMT but fails to transfer semantics to organic inputs, showing overfitting to template constraints.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.06349","ref_index":77,"ref_count":1,"confidence":0.88,"is_internal_anchor":false,"paper_title":"\"Chi nas dal soch el sent de legn\" -- Auditing Text Corpora for Lombard","primary_cat":"cs.CL","submitted_at":"2026-06-04T16:20:14+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"Manual audit shows web-scraped Lombard corpora are largely noisy and biased toward Western varieties over Eastern ones.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.01800","ref_index":22,"ref_count":1,"confidence":0.88,"is_internal_anchor":false,"paper_title":"Multilinguality of Large Language Models From a Structural Perspective","primary_cat":"cs.CL","submitted_at":"2026-06-01T07:18:09+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"Low-resource languages are structurally more different from English in LLMs than high- or mid-resource ones, and language-specific post-training alters structures while preserving inter-language relationships.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.01136","ref_index":12,"ref_count":1,"confidence":0.88,"is_internal_anchor":false,"paper_title":"From Outliers to Errors: Auditing Pali-to-English LLM Translations with Multi-Reference Adjudication","primary_cat":"cs.CL","submitted_at":"2026-05-31T10:15:36+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"A multi-reference audit framework for LLM translations of the Pali Canon uses embedding drift from a human reference centroid to triage candidates for LLM-judge adjudication, showing drift correlates with major error rates and model-specific differences in the high-drift tail.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.00356","ref_index":55,"ref_count":1,"confidence":0.88,"is_internal_anchor":false,"paper_title":"How Far Do Auto-Interpretation Labels Generalize: A Controlled Study Across Languages, Scripts, and Rewordings","primary_cat":"cs.CL","submitted_at":"2026-05-29T20:59:28+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":7.0,"formal_verification":"none","one_line_summary":"Auto-interpretation labels for SAE features generalize poorly across languages and scripts, missing the same semantic content up to 4x more often in Serbian than English and more in Cyrillic than Latin despite deterministic transliteration.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2605.28163","ref_index":15,"ref_count":1,"confidence":0.88,"is_internal_anchor":false,"paper_title":"DEPART: DEcomposing PARiTy across Multilingual LLMs","primary_cat":"cs.CL","submitted_at":"2026-05-27T08:45:48+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"A Bayesian framework decomposes mLLM variance, showing language features explain 79-92% of language identity variance and that model identity vs. benchmark-model interactions dominate differently for understanding versus reasoning tasks.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2605.25846","ref_index":13,"ref_count":1,"confidence":0.88,"is_internal_anchor":false,"paper_title":"On the Limits of Model Merging for Multilinguality in Pre-Training","primary_cat":"cs.CL","submitted_at":"2026-05-25T13:38:33+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":5.0,"formal_verification":"none","one_line_summary":"Merging any combination of monolingual pre-trained models leads to performance collapse due to interference, indicating that merging flexibility from fine-tuning does not extend to pre-training.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2605.22821","ref_index":6,"ref_count":1,"confidence":0.88,"is_internal_anchor":false,"paper_title":"Tokenisation via Convex Relaxations","primary_cat":"cs.CL","submitted_at":"2026-05-21T17:59:56+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":7.0,"formal_verification":"none","one_line_summary":"ConvexTok uses convex relaxation of tokenization to a linear program, improving intrinsic metrics, bits-per-byte, and some downstream tasks while certifying near-optimality within 1% at typical vocabulary sizes.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2605.13050","ref_index":13,"ref_count":2,"confidence":0.88,"is_internal_anchor":false,"paper_title":"Context Training with Active Information Seeking","primary_cat":"cs.CL","submitted_at":"2026-05-13T06:15:32+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":5.0,"formal_verification":"none","one_line_summary":"Active information seeking via search tools, when combined with multi-candidate context pruning during training, produces consistent gains on translation, health, and reasoning tasks over naive tool addition or no-tool baselines.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2604.08977","ref_index":19,"ref_count":1,"confidence":0.88,"is_internal_anchor":false,"paper_title":"Testing the Assumptions of Active Learning for Translation Tasks with Few Samples","primary_cat":"cs.CL","submitted_at":"2026-04-10T05:30:25+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":5.0,"formal_verification":"none","one_line_summary":"Informativeness and diversity of samples selected by active learning show no correlation with test performance on translation tasks using few samples; ordering and pre-training effects dominate instead.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2604.02596","ref_index":20,"ref_count":1,"confidence":0.88,"is_internal_anchor":false,"paper_title":"An Empirical Study of Many-Shot In-Context Learning for Machine Translation of Low-Resource Languages","primary_cat":"cs.CL","submitted_at":"2026-04-03T00:13:34+00:00","verdict":"ACCEPT","verdict_confidence":"HIGH","novelty_score":5.0,"formal_verification":"none","one_line_summary":"BM25-retrieved many-shot examples match much larger random sets for translating English into ten truly low-resource languages, and ICL still helps after fine-tuning.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null}],"limit":50,"offset":0}