{
  "id": 5773,
  "url": "https://arxiv.org/abs/2604.15038v2",
  "title": "When Fairness Metrics Disagree: Evaluating the Reliability of Demographic Fairness Assessment in Machine Learning",
  "summary": "The evaluation of fairness in machine learning systems has become a central concern in high-stakes applications, including biometric recognition, healthcare decision-making, and automated risk assessment. Existing approaches typically rely on a small number of fairness metrics to assess model behaviour across group partitions, implicitly assuming that these metrics provide consistent and reliable conclusions. However, different fairness metrics capture distinct statistical properties of model pe",
  "authors": "Khalid Adnan Alsayed",
  "category": "research",
  "topics": "bias-fairness,privacy-surveillance,healthcare",
  "orgs": null,
  "regions": null,
  "published_at": "2026-04-16T14:07:37.000Z",
  "fetched_at": "2026-07-14T16:32:02.057Z",
  "source_slug": "arxiv-ethics",
  "source_name": "arXiv",
  "source_homepage": "https://arxiv.org",
  "ethics_ai_record_url": "https://ethics.ai/record/5773",
  "original_url": "https://arxiv.org/abs/2604.15038v2",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}