{
  "id": 14470,
  "url": "https://arxiv.org/abs/2607.25375v1",
  "title": "Inspect India Evals: An Open Benchmarking Framework for Evaluating Large Language Models in the Indian Linguistic and Cultural Context",
  "summary": "India is a vast nation of over 1.4 billion people, varied by hundreds of diverse and locally specific traditions and cultures and 22 officially recognized languages. Large language models (LLMs) are now being deployed on a massive scale throughout the mainland as well as in remote villages. However, the common benchmarks - MMLU, BIG-Bench, and TruthfulQA are almost exclusively English- and Western-centric. They do not identify those safety, fairness, and accuracy failures unique to the Indian co",
  "authors": "Abhishek Kumar Singh, Shrey Nag, Sachita, Lipi Goel, Rajeshwar Singh Janwar",
  "category": "research",
  "topics": "bias-fairness",
  "orgs": null,
  "regions": "india",
  "published_at": "2026-07-28T07:30:12.000Z",
  "fetched_at": "2026-07-29T05:10:12.205Z",
  "source_slug": "x-arxiv-red-teaming-query",
  "source_name": "arXiv red teaming query",
  "source_homepage": "https://arxiv.org/a/redteam",
  "ethics_ai_record_url": "https://ethics.ai/record/14470",
  "original_url": "https://arxiv.org/abs/2607.25375v1",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}