{
  "id": 12213,
  "url": "https://arxiv.org/abs/2607.18068v1",
  "title": "Human Grounded Evaluation of Large Language Models for Optical Network Automation",
  "summary": "Large language models (LLMs) are increasingly adopted for network automation, yet their output quality and inference cost can vary substantially across LLM families. We present HuGLEN, a stepwise evaluation pipeline that uses an LLM-as-a-judge together with a small set of expert ratings to enable scalable and reproducible comparison of candidate LLMs, and to rank them using a quality efficiency score (QES). We demonstrate HuGLEN for translating outputs from an explainable artificial intelligence",
  "authors": "Kiarash Rezaei, Omran Ayoub, Paolo Monti, Carlos Natalino",
  "category": "research",
  "topics": "jobs-economy,transparency",
  "orgs": null,
  "regions": null,
  "published_at": "2026-07-20T15:36:00.000Z",
  "fetched_at": "2026-07-21T05:10:12.656Z",
  "source_slug": "x-arxiv-cs-ai",
  "source_name": "arXiv cs.AI",
  "source_homepage": "https://arxiv.org/list/cs.AI/recent",
  "ethics_ai_record_url": "https://ethics.ai/record/12213",
  "original_url": "https://arxiv.org/abs/2607.18068v1",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}