{
  "id": 3721,
  "url": "https://arxiv.org/abs/2605.26315v1",
  "title": "Curriculum Learning for Safety Alignment",
  "summary": "Direct Preference Optimisation (DPO) is widely used for safety alignment in large language models. However, prior work shows it is brittle and exhibits poor out-of-distribution (OOD) generalisation. In this paper, we investigate whether Curriculum Learning can improve the robustness of DPO-based safety alignment. We propose Staged-Competence, a curriculum-based framework that organises preference data by difficulty, employs competence-based sampling, and progressively updates the reference model",
  "authors": "Sandeep Kumar, Virginia Smith, Chhavi Yadav",
  "category": "research",
  "topics": "safety-alignment,finance-investment",
  "orgs": null,
  "regions": null,
  "published_at": "2026-05-25T20:13:06.000Z",
  "fetched_at": "2026-07-14T16:30:27.610Z",
  "source_slug": "arxiv-ethics",
  "source_name": "arXiv",
  "source_homepage": "https://arxiv.org",
  "ethics_ai_record_url": "https://ethics.ai/record/3721",
  "original_url": "https://arxiv.org/abs/2605.26315v1",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}