{
  "id": 5523,
  "url": "https://arxiv.org/abs/2604.19998v1",
  "title": "What Makes a Good AI Review? Concern-Level Diagnostics for AI Peer Review",
  "summary": "Evaluating AI-generated reviews by verdict agreement is widely recognized as insufficient, yet current alternatives rarely audit which concerns a system identifies, how it prioritizes them, or whether those priorities align with the review rationale that shaped the final assessment. We propose concern alignment, a diagnostic framework that evaluates AI reviews at the concern level rather than only at the verdict level. The framework's core data structure is the match graph, a bipartite alignment",
  "authors": "Ming Jin",
  "category": "research",
  "topics": "safety-alignment,healthcare,transparency",
  "orgs": null,
  "regions": null,
  "published_at": "2026-04-21T21:16:59.000Z",
  "fetched_at": "2026-07-14T16:31:48.875Z",
  "source_slug": "arxiv-ethics",
  "source_name": "arXiv",
  "source_homepage": "https://arxiv.org",
  "ethics_ai_record_url": "https://ethics.ai/record/5523",
  "original_url": "https://arxiv.org/abs/2604.19998v1",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}