{"total":16,"items":[{"citing_arxiv_id":"2607.05679","ref_index":23,"ref_count":1,"confidence":0.88,"is_internal_anchor":false,"paper_title":"RPAM: A Principled Metric for Evaluating Associations in Language Models with High Predictive Validity in Downstream Outputs","primary_cat":"cs.CL","submitted_at":"2026-07-06T22:48:13+00:00","verdict":"CONDITIONAL","verdict_confidence":"MODERATE","novelty_score":5.5,"formal_verification":"none","one_line_summary":"Relative Probability Association Metric (RPAM) measures LM associations via softmax-normalized continuation probabilities and correlates strongly with human associations and downstream LM behavior across three models.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.28063","ref_index":47,"ref_count":1,"confidence":0.88,"is_internal_anchor":false,"paper_title":"How to deal with machine learning bias in economic history","primary_cat":"econ.GN","submitted_at":"2026-06-26T13:10:11+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":5.0,"formal_verification":"none","one_line_summary":"The paper guides ML use in economic history, identifies systematic prediction bias that distorts coefficients, and shows debiasing via small expert-labeled samples can correct it while preserving scale.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.24022","ref_index":12,"ref_count":1,"confidence":0.88,"is_internal_anchor":false,"paper_title":"Do Language Models Pass the Bechdel Test? Auditing Gender Biases in LLM-Generated Screenplays","primary_cat":"cs.HC","submitted_at":"2026-06-23T00:00:12+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":4.0,"formal_verification":"none","one_line_summary":"Human-written screenplays pass the Bechdel test more often than those generated by GPT-5, Gemini 3 Pro, and Claude Sonnet 4.5, though network analyses show mixed bias patterns across all script types.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.13755","ref_index":1,"ref_count":1,"confidence":0.88,"is_internal_anchor":false,"paper_title":"Position: Align AI to Our Aspirations, Not Our Flaws","primary_cat":"cs.CY","submitted_at":"2026-06-11T16:03:46+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":4.0,"formal_verification":"none","one_line_summary":"AI alignment should target objective floors of competence, accuracy, honesty, and lawfulness rather than aggregated human preferences.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.12088","ref_index":102,"ref_count":1,"confidence":0.88,"is_internal_anchor":false,"paper_title":"Debiasing Without Protected Attributes: Latent Concept Erasure from Textual Profiles","primary_cat":"cs.CL","submitted_at":"2026-06-10T13:49:27+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"H-SAL erases latent concepts from text profiles using self-descriptions as implicit debiasing signals and shows competitive performance on a new multi-domain Stack Exchange helpfulness benchmark.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.02776","ref_index":6,"ref_count":1,"confidence":0.88,"is_internal_anchor":false,"paper_title":"Topics as Proxies for Sociodemographics: How Conversational Context Affects LLM Answers","primary_cat":"cs.CL","submitted_at":"2026-06-01T18:38:41+00:00","verdict":null,"verdict_confidence":null,"novelty_score":null,"formal_verification":null,"one_line_summary":null,"context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2606.00600","ref_index":6,"ref_count":1,"confidence":0.88,"is_internal_anchor":false,"paper_title":"Understanding the Self-Reflection Mechanisms of LLMs through Biased Attitude Associations","primary_cat":"cs.SI","submitted_at":"2026-05-30T07:57:56+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":4.0,"formal_verification":"none","one_line_summary":"ReBias-Lens shows LLM self-reflection produces layer-wise smoothing of global valence fluctuations that reduces behavioral bias overall, yet selectively locks in and amplifies certain category-specific biases.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2605.11672","ref_index":3,"ref_count":1,"confidence":0.88,"is_internal_anchor":false,"paper_title":"A CAP-like Trilemma for Large Language Models: Correctness, Non-bias, and Utility under Semantic Underdetermination","primary_cat":"cs.AI","submitted_at":"2026-05-12T07:28:38+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"Under semantic underdetermination, LLMs cannot always guarantee strong correctness, strict non-bias, and high utility at once.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2605.09647","ref_index":36,"ref_count":1,"confidence":0.88,"is_internal_anchor":false,"paper_title":"Modeling Implicit Conflict Monitoring Mechanisms against Stereotypes in LLMs","primary_cat":"cs.SI","submitted_at":"2026-05-10T16:46:26+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"LLMs contain identifiable COCO neurons that enable implicit self-correction against stereotypes; targeted editing of these neurons improves fairness and robustness to jailbreaks while preserving generation quality.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2605.01048","ref_index":5,"ref_count":1,"confidence":0.88,"is_internal_anchor":false,"paper_title":"Compared to What? Baselines and Metrics for Counterfactual Prompting","primary_cat":"cs.CL","submitted_at":"2026-05-01T19:23:33+00:00","verdict":"CONDITIONAL","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"Counterfactual prompting effects on LLMs are often indistinguishable from those caused by meaning-preserving paraphrases, causing most previously reported demographic sensitivities to disappear under proper statistical comparison.","context_count":1,"top_context_role":"other","top_context_polarity":"unclear","context_text":"Let di ∈ {+ 1, −1, 0} denote a signed perturbation indicator: +1 or −1 for the two perturbation directions, and 0 for the baseline condition. We consider two complementary specifications. Thedifference modelpairs each perturbed observation with its baseline, so the baseline is absorbed into the difference andd i ∈ {+1,−1}: ∆i =β pert ·d i +ε i,∆ i =y perturbed,i −y baseline,i (5) This yields a coefficient ˆβpert that estimates the directional shift associated with the perturba- tion. We also consider a complementarylevel modelthat includes the original-prompt output as a covariate (Appendix A); in our experiments, both specifications yield nearly identical estimates. The effect magnitude is | ˆβpert|; the direction is assessed through a one-sided"},{"citing_arxiv_id":"2604.17398","ref_index":4,"ref_count":1,"confidence":0.88,"is_internal_anchor":false,"paper_title":"Contrastive Analysis of Linguistic Representations in Large Language Model Outputs through Structured Synthetic Data Generation and Abstracted N-gram Associations","primary_cat":"cs.CL","submitted_at":"2026-04-19T12:02:17+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"A methodological framework detects subtle group-associated linguistic biases in LLM outputs by generating controlled synthetic minimal pairs, abstracting n-grams, and ranking high-signal fragments with a PMI variant for expert review.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2503.11572","ref_index":20,"ref_count":1,"confidence":0.88,"is_internal_anchor":false,"paper_title":"Implicit Bias-Like Patterns in Reasoning Models","primary_cat":"cs.CY","submitted_at":"2025-03-14T16:40:02+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"Reasoning models expend more tokens on association-incompatible tasks than compatible ones, indicating greater effort on counter-stereotypical information, except for Claude 3.7 Sonnet which shows the reverse pattern linked to its bias-focused reasoning.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2112.04359","ref_index":42,"ref_count":1,"confidence":0.88,"is_internal_anchor":false,"paper_title":"Ethical and social risks of harm from Language Models","primary_cat":"cs.CL","submitted_at":"2021-12-08T16:09:48+00:00","verdict":"ACCEPT","verdict_confidence":"MODERATE","novelty_score":6.0,"formal_verification":"none","one_line_summary":"The authors provide a detailed taxonomy of 21 risks associated with language models, covering discrimination, information leaks, misinformation, malicious applications, interaction harms, and societal impacts like job loss and environmental costs.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"2110.08193","ref_index":5,"ref_count":1,"confidence":0.88,"is_internal_anchor":false,"paper_title":"BBQ: A Hand-Built Bias Benchmark for Question Answering","primary_cat":"cs.CL","submitted_at":"2021-10-15T16:43:46+00:00","verdict":"ACCEPT","verdict_confidence":"LOW","novelty_score":7.0,"formal_verification":"none","one_line_summary":"BBQ is a new benchmark dataset showing that QA models often default to social stereotypes, achieving up to 3.4 points higher accuracy when the correct answer aligns with bias.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"1906.10256","ref_index":6,"ref_count":1,"confidence":0.88,"is_internal_anchor":false,"paper_title":"Good Secretaries, Bad Truck Drivers? Occupational Gender Stereotypes in Sentiment Analysis","primary_cat":"cs.CL","submitted_at":"2019-06-24T22:31:33+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"Authors release a new 800-sentence gender-balanced profession dataset and use it to test occupational gender stereotypes in three sentiment analysis models.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null},{"citing_arxiv_id":"1906.10007","ref_index":5,"ref_count":1,"confidence":0.88,"is_internal_anchor":false,"paper_title":"Language Modelling Makes Sense: Propagating Representations through WordNet for Full-Coverage Word Sense Disambiguation","primary_cat":"cs.CL","submitted_at":"2019-06-24T14:59:12+00:00","verdict":"UNVERDICTED","verdict_confidence":"LOW","novelty_score":6.0,"formal_verification":"none","one_line_summary":"Contextual embeddings are propagated through WordNet to produce full-coverage sense representations that let a simple k-NN classifier outperform prior neural WSD models.","context_count":0,"top_context_role":null,"top_context_polarity":null,"context_text":null}],"limit":50,"offset":0}