{
  "id": 428,
  "url": "https://arxiv.org/abs/2607.00048v1",
  "title": "Comparing Large Language Models on Scrum Certification-Style Questions: Accuracy, Stability, and Error Patterns",
  "summary": "Large Language Models (LLMs) are increasingly used in exam- and certification-style question answering tasks, where their ability to retrieve, interpret, and apply domain-specific knowledge can be systematically assessed. In Software Engineering, such settings are particularly relevant when questions depend on strict adherence to normative definitions, roles, artifacts, and rules. This paper evaluates the performance of three contemporary LLMs, \\textit{GPT-5 mini}, \\textit{Gemini 3 Flash}, and \\",
  "authors": "Robson Alves Vilar, Emanuel Dantas Filho, Ademar França de Sousa Neto, Mirko Perkusich, Danyllo Wagner Albuquerque, João Paiva et al.",
  "category": "research",
  "topics": null,
  "orgs": "openai,google",
  "regions": null,
  "published_at": "2026-06-29T23:37:56.000Z",
  "fetched_at": "2026-07-14T14:14:32.646Z",
  "source_slug": "arxiv-ethics",
  "source_name": "arXiv",
  "source_homepage": "https://arxiv.org",
  "ethics_ai_record_url": "https://ethics.ai/record/428",
  "original_url": "https://arxiv.org/abs/2607.00048v1",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}