{
  "id": 788,
  "url": "https://arxiv.org/abs/2606.20814v1",
  "title": "What Shapes Emergent Misalignment? Insights from Training Dynamics, Model Priors, and Data",
  "summary": "Emergent misalignment (EM) is a phenomenon in which models generalize with narrow fine-tuning, leading to broad (yet uneven) misalignment across evaluation questions. We study EM and its variability directly through the components of fine-tuning: training dynamics, model priors, and data. (1) We first explored how in-domain training loss relates to out-of-domain alignment scores across datasets and model families. Then, we tried to induce potential alternative local minima through different lear",
  "authors": "Yuchen Zhang, Anietta Weckauff, Diego Garcia-Olano, Maksym Andriushchenko",
  "category": "research",
  "topics": "safety-alignment",
  "orgs": null,
  "regions": null,
  "published_at": "2026-06-18T18:04:20.000Z",
  "fetched_at": "2026-07-14T14:14:46.036Z",
  "source_slug": "arxiv-ethics",
  "source_name": "arXiv",
  "source_homepage": "https://arxiv.org",
  "ethics_ai_record_url": "https://ethics.ai/record/788",
  "original_url": "https://arxiv.org/abs/2606.20814v1",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}