{
  "id": 4904,
  "url": "https://arxiv.org/abs/2605.05003v1",
  "title": "Misaligned by Reward: Socially Undesirable Preferences in LLMs",
  "summary": "Reward models are a key component of large language model alignment, serving as proxies for human preferences during training. However, existing evaluations focus primarily on broad instruction-following benchmarks, providing limited insight into whether these models capture socially desirable preferences. As a result, important failures in social alignment can remain hidden. We extend reward-model benchmarking to four socially consequential domains: bias, safety, morality, and ethical reasoning",
  "authors": "Gayane Ghazaryan, Esra Dönmez",
  "category": "research",
  "topics": "bias-fairness,safety-alignment",
  "orgs": null,
  "regions": null,
  "published_at": "2026-05-06T15:04:23.000Z",
  "fetched_at": "2026-07-14T16:31:21.932Z",
  "source_slug": "arxiv-ethics",
  "source_name": "arXiv",
  "source_homepage": "https://arxiv.org",
  "ethics_ai_record_url": "https://ethics.ai/record/4904",
  "original_url": "https://arxiv.org/abs/2605.05003v1",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}