{
  "id": 78,
  "url": "https://arxiv.org/abs/2607.08842v1",
  "title": "L2-Bench: An Evaluation Benchmark for Measuring LLM Capabilities in Second Language Education",
  "summary": "Despite rapid AI adoption in education, rigorous evaluation of AI-powered educational (AIED) systems remains critically underdeveloped, particularly in second language (L2) education, one of the most common yet least evaluated AI applications. We introduce L2-Bench, an open-source benchmark of 1,000+ task-response pairs to aid the pedagogy-led evaluation of LLM capabilities relating to language learning and assessment. Crucially, L2-Bench measures model performativity on the application of learn",
  "authors": "James Edgell, Wm. Matthew Kennedy, Ben Knight, Danielle Carvalho, Martin Ku",
  "category": "research",
  "topics": "children-education",
  "orgs": null,
  "regions": null,
  "published_at": "2026-07-09T18:02:55.000Z",
  "fetched_at": "2026-07-14T14:14:15.666Z",
  "source_slug": "arxiv-ethics",
  "source_name": "arXiv",
  "source_homepage": "https://arxiv.org",
  "ethics_ai_record_url": "https://ethics.ai/record/78",
  "original_url": "https://arxiv.org/abs/2607.08842v1",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}