{
  "id": 15343,
  "url": "https://blog.redwoodresearch.org/p/sota-alignment-assessments-dont-strongly",
  "title": "SOTA alignment assessments don’t strongly update us against misalignment",
  "summary": "Anthropic concluded in the April Mythos Preview alignment risk update that the model “does not possess any unknown propensities that would increase alignment risk.” The report argues that if Mythos Preview were coherently misaligned[1][2], it likely would have been detected by the assessment (following Anthropic, I will call this “reliability of the assessment”",
  "authors": "Alexa Pan",
  "category": "org",
  "topics": "safety-alignment",
  "orgs": "anthropic",
  "regions": null,
  "published_at": "2026-07-31T23:09:20.000Z",
  "fetched_at": "2026-08-01T05:10:57.676Z",
  "source_slug": "redwood",
  "source_name": "Redwood Research",
  "source_homepage": "https://blog.redwoodresearch.org",
  "ethics_ai_record_url": "https://ethics.ai/record/15343",
  "original_url": "https://blog.redwoodresearch.org/p/sota-alignment-assessments-dont-strongly",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}