{
  "id": 18362,
  "url": "https://arxiv.org/abs/2608.10268",
  "title": "Toward Human Rights Benchmarking for LLMs: A Pilot Methodology",
  "summary": "arXiv:2608.10268v1 Announce Type: cross Abstract: Large language models (LLMs) increasingly mediate legal determinations over what human rights are realized, and how. Yet, no evaluation benchmark exists to assess whether they can reason correctly about human rights law. To this end, we report our efforts to develop a robust and scalable methodology for creating HumRightsBench: the first expert-validated, scenario-based benchmark for evaluating reasoning grounded in the obligation structure of in",
  "authors": "Savannah Thais, Wm. Matthew Kennedy, Abhigyan Acherjee, Matilda Wysocki, Malcolm Langford, Caitlin Kraft Buchman",
  "category": "research",
  "topics": "regulation",
  "orgs": null,
  "regions": null,
  "published_at": "2026-08-12T04:00:00.000Z",
  "fetched_at": "2026-08-12T05:10:43.828Z",
  "source_slug": "arxiv-cscy",
  "source_name": "arXiv cs.CY",
  "source_homepage": "https://arxiv.org/list/cs.CY/recent",
  "ethics_ai_record_url": "https://ethics.ai/record/18362",
  "original_url": "https://arxiv.org/abs/2608.10268",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}