{
  "id": 3000,
  "url": "https://arxiv.org/abs/2607.11871v1",
  "title": "Inside the Unfair Judge: A Mechanistic Interpretability Account of LLM-as-Judge Bias",
  "summary": "Existing studies of LLM-as-judge scoring bias work predominantly at the input-output level: they perturb inputs, measure score deltas, and propose prompt-level mitigations. We argue that the same biases admit a representation-level account in the judge's hidden state, complementary to the input-output view and operationally useful in ways it does not afford. We report three findings, across seven judges, seven bias types, and nine benchmarks. Geometry: baseline judging inputs occupy a tight acti",
  "authors": "Zixiang Xu, Sixian Li, Huaxing Liu, Xiang Wang, Shuai Li, Zirui Song, Xiuying Chen",
  "category": "research",
  "topics": "bias-fairness,safety-alignment",
  "orgs": null,
  "regions": null,
  "published_at": "2026-07-13T17:55:19.000Z",
  "fetched_at": "2026-07-14T16:11:46.979Z",
  "source_slug": "x-arxiv-cs-ai",
  "source_name": "arXiv cs.AI",
  "source_homepage": "https://arxiv.org/list/cs.AI/recent",
  "ethics_ai_record_url": "https://ethics.ai/record/3000",
  "original_url": "https://arxiv.org/abs/2607.11871v1",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}