{
  "id": 3414,
  "url": "https://arxiv.org/abs/2606.00334v1",
  "title": "Isolating LLM Lexical Bias: A Curation-Free Triangulated Metric for Preference-Stage Learning",
  "summary": "Various language domains have undergone remarkable changes in recent years; these shifts are largely attributed to the advent of Large Language Models and their misalignment with natural language usage. These misalignments are thought to partly originate in the preference-learning stage, e.g. Reinforcement Learning from Human Feedback, which generally makes models more useful but simultaneously may introduce systematic lexical bias. In terms of lexical behavior, this is visible in a model's pref",
  "authors": "Xiaoyang Ming, Jose Hernandez, Thomas Stephan Juzek",
  "category": "research",
  "topics": "bias-fairness,safety-alignment",
  "orgs": null,
  "regions": null,
  "published_at": "2026-05-29T20:19:49.000Z",
  "fetched_at": "2026-07-14T16:30:14.369Z",
  "source_slug": "arxiv-ethics",
  "source_name": "arXiv",
  "source_homepage": "https://arxiv.org",
  "ethics_ai_record_url": "https://ethics.ai/record/3414",
  "original_url": "https://arxiv.org/abs/2606.00334v1",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}