{
  "id": 16154,
  "url": "https://arxiv.org/abs/2608.01423v1",
  "title": "Scoring Rules! Statistical and Strategic Alignment for Text Evaluation Metrics",
  "summary": "Reference-based text evaluation metrics, which are widely used to assess natural language generation systems, score a candidate response by comparing it with a reference response. The reliability of an evaluation metric is usually judged by its statistical correlation with human ratings. However, as these metrics are increasingly used as optimization objectives, correlation alone is no longer sufficient: agents may strategically game the evaluation metric. We study this issue through two complem",
  "authors": "Shengwei Xu, Yuxuan Lu, Yifan Wu, Jason Hartline, Grant Schoenebeck",
  "category": "research",
  "topics": "safety-alignment,agents-autonomy",
  "orgs": null,
  "regions": null,
  "published_at": "2026-08-02T18:10:01.000Z",
  "fetched_at": "2026-08-04T05:10:21.797Z",
  "source_slug": "arxiv-cslg",
  "source_name": "arXiv cs.LG",
  "source_homepage": "https://arxiv.org/list/cs.LG/recent",
  "ethics_ai_record_url": "https://ethics.ai/record/16154",
  "original_url": "https://arxiv.org/abs/2608.01423v1",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}