{
  "id": 4660,
  "url": "https://arxiv.org/abs/2605.10998v1",
  "title": "Few-Shot Truly Benign DPO Attack for Jailbreaking LLMs",
  "summary": "Fine-tuning APIs make frontier LLMs easy to customize, but they can also weaken safety alignment during fine-tuning. While prior work shows that benign supervised fine-tuning (SFT) can reduce refusal behavior, deployed fine-tuning pipelines increasingly support preference-based objectives, whose safety risks remain less understood. We show that Direct Preference Optimization (DPO) introduces a stronger and harder-to-audit failure mode. We propose a truly benign DPO attack using only 10 harmless ",
  "authors": "Sangyeon Yoon, Wonje Jeung, Yoonjun Cho, Dongjae Jeon, Albert No",
  "category": "research",
  "topics": "safety-alignment,transparency",
  "orgs": null,
  "regions": null,
  "published_at": "2026-05-09T15:52:29.000Z",
  "fetched_at": "2026-07-14T16:31:08.358Z",
  "source_slug": "arxiv-ethics",
  "source_name": "arXiv",
  "source_homepage": "https://arxiv.org",
  "ethics_ai_record_url": "https://ethics.ai/record/4660",
  "original_url": "https://arxiv.org/abs/2605.10998v1",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}