{
  "id": 18681,
  "url": "https://arxiv.org/abs/2608.10126v1",
  "title": "Procedural Fairness Failures in RLHF from Preference Averaging",
  "summary": "Reinforcement Learning from Human Feedback (RLHF) aggregates heterogeneous preferences into a single reward model, assuming preference homogeneity. When preferences are heterogeneous, this aggregation induces a procedural fairness failure where majority preference groups dominate reward learning while minority preferences are systematically under-represented. This work defines procedural fairness in alignment as preserving distinct preference signals during reward modeling and shows that standar",
  "authors": "M P V S Gopinadh, Karthik Kamuju, Kummari Avinash, John Joshua, Srinivasa Raju Rudraraju",
  "category": "research",
  "topics": "bias-fairness,safety-alignment",
  "orgs": null,
  "regions": null,
  "published_at": "2026-08-10T18:38:16.000Z",
  "fetched_at": "2026-08-12T05:10:43.828Z",
  "source_slug": "x-arxiv-fairness-query",
  "source_name": "arXiv fairness query",
  "source_homepage": "https://arxiv.org",
  "ethics_ai_record_url": "https://ethics.ai/record/18681",
  "original_url": "https://arxiv.org/abs/2608.10126v1",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}