{
  "id": 333,
  "url": "https://arxiv.org/abs/2607.01740v1",
  "title": "Meta-Benchmarks for Financial-Services LLM Evaluation",
  "summary": "Public LLM leaderboards optimise for global average performance and do not capture the specific cognitive demands of financial-services work: a model that leads on MMLU-Pro may underperform on document-grounded compliance reasoning, and a coding leader may handle multi-turn customer interactions poorly. We present a meta-benchmarking framework that organises 452 publicly reported benchmarks into 41 O*NET Generalized Work Activities and aggregates those into 38 BIAN banking business domains spann",
  "authors": "Blair Hudson",
  "category": "research",
  "topics": "regulation,finance-investment",
  "orgs": "meta",
  "regions": null,
  "published_at": "2026-07-02T05:52:32.000Z",
  "fetched_at": "2026-07-14T14:14:28.435Z",
  "source_slug": "arxiv-ethics",
  "source_name": "arXiv",
  "source_homepage": "https://arxiv.org",
  "ethics_ai_record_url": "https://ethics.ai/record/333",
  "original_url": "https://arxiv.org/abs/2607.01740v1",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}