{
  "id": 7097,
  "url": "https://arxiv.org/abs/2603.17044v2",
  "title": "Do Understanding and Generation Fight? A Diagnostic Study of DPO for Unified Multimodal Models",
  "summary": "Unified multimodal models share a language model backbone for both understanding and generating images. Can DPO align both capabilities simultaneously? We present the first systematic study of this question, applying DPO to Janus-Pro at 1B and 7B parameters under seven training strategies and two post-hoc methods. The central finding is negative: generation quality resists DPO alignment across all tested conditions on this architecture. No method improves generation CLIPScore at 7B (|Delta| 0.5 ",
  "authors": "Abinav Rao, Sujan Rachuri",
  "category": "research",
  "topics": "safety-alignment,healthcare",
  "orgs": null,
  "regions": null,
  "published_at": "2026-03-17T18:26:29.000Z",
  "fetched_at": "2026-07-14T16:32:59.164Z",
  "source_slug": "arxiv-ethics",
  "source_name": "arXiv",
  "source_homepage": "https://arxiv.org",
  "ethics_ai_record_url": "https://ethics.ai/record/7097",
  "original_url": "https://arxiv.org/abs/2603.17044v2",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}