{
  "id": 3609,
  "url": "https://arxiv.org/abs/2605.28183v3",
  "title": "BenGER: Benchmarking LLM Systems on Subsumption-Based Legal Reasoning in German Law",
  "summary": "We introduce BenGER (Benchmark for German Law), a benchmark and dataset for evaluating LLM systems on subsumption-based legal reasoning in German law. The dataset combines 596 exam-style free-text legal case tasks across multiple levels of legal education and 531 short doctrinal reasoning tasks. It includes a controlled validation subset of timed human-written solutions under both unaided and human-AI co-creation conditions. We evaluate 12 contemporary LLM systems - closed flagship, efficiency-o",
  "authors": "Sebastian Nagl, Ann-Kristin Mayrhofer, Martin Heidebach, Aleyna Koçak, Anne Zettelmeier, Elly Breu et al.",
  "category": "research",
  "topics": "regulation,children-education",
  "orgs": null,
  "regions": null,
  "published_at": "2026-05-27T09:03:59.000Z",
  "fetched_at": "2026-07-14T16:30:23.246Z",
  "source_slug": "arxiv-ethics",
  "source_name": "arXiv",
  "source_homepage": "https://arxiv.org",
  "ethics_ai_record_url": "https://ethics.ai/record/3609",
  "original_url": "https://arxiv.org/abs/2605.28183v3",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}