{
  "id": 15418,
  "url": "https://simonwillison.net/2026/Jul/31/smevals",
  "title": "smevals - a small eval suite for evaluating models, prompts, and harnesses",
  "summary": "smevals - a small eval suite for evaluating models, prompts, and harnesses I've been working with Jesse Vincent's Prime Radiant applied AI research lab building out this evals framework to help answer questions about the capabilities of different models. The result is smevals , a new tool for running small eval suites across different model configurations and grading the results. The blog entry describes the tool in detail. Here's the 10 second version: Tell your coding agent to run uvx smevals",
  "authors": null,
  "category": "org",
  "topics": "agents-autonomy",
  "orgs": null,
  "regions": null,
  "published_at": "2026-07-31T21:15:23.000Z",
  "fetched_at": "2026-08-01T05:10:57.676Z",
  "source_slug": "x-simon-willisons-weblog",
  "source_name": "Simon Willisons Weblog",
  "source_homepage": "https://simonwillison.net",
  "ethics_ai_record_url": "https://ethics.ai/record/15418",
  "original_url": "https://simonwillison.net/2026/Jul/31/smevals",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}