{
  "id": 16275,
  "url": "https://arxiv.org/abs/2608.03659v1",
  "title": "How Closely Do LLM Reviews Align with Human Peer Review?",
  "summary": "Large language models (LLMs) are increasingly used to generate scientific reviews, yet existing evaluations rarely examine whether different providers align with both conference decisions and human reviewing priorities within the same controlled setting. We compare reviews from OpenAI GPT-5.4, Google Gemini 3.1 Pro Preview, and Anthropic Claude Opus 4.6 with human reviews and final decisions for 300 topic-matched ICLR 2026 submissions, equally divided among oral, poster, and rejected papers. Eac",
  "authors": "Abraham Camelo-Guerrero, Jairo Diaz-Rodriguez",
  "category": "research",
  "topics": null,
  "orgs": "openai,anthropic,google",
  "regions": null,
  "published_at": "2026-08-04T13:39:36.000Z",
  "fetched_at": "2026-08-05T05:10:44.550Z",
  "source_slug": "arxiv-ethics",
  "source_name": "arXiv",
  "source_homepage": "https://arxiv.org",
  "ethics_ai_record_url": "https://ethics.ai/record/16275",
  "original_url": "https://arxiv.org/abs/2608.03659v1",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}