{
  "id": 6618,
  "url": "https://arxiv.org/abs/2603.27693v1",
  "title": "LVRPO: Language-Visual Alignment with GRPO for Multimodal Understanding and Generation",
  "summary": "Unified multimodal pretraining has emerged as a promising paradigm for jointly modeling language and vision within a single foundation model. However, existing approaches largely rely on implicit or indirect alignment signals and remain suboptimal for simultaneously supporting multimodal understanding and generation, particularly in settings that require fine-grained language-visual reasoning and controllable generation. In this work, we propose LVRPO, a language-visual reinforcement-based prefe",
  "authors": "Shentong Mo, Sukmin Yun",
  "category": "research",
  "topics": "safety-alignment",
  "orgs": null,
  "regions": null,
  "published_at": "2026-03-29T13:38:21.000Z",
  "fetched_at": "2026-07-14T16:32:37.309Z",
  "source_slug": "arxiv-ethics",
  "source_name": "arXiv",
  "source_homepage": "https://arxiv.org",
  "ethics_ai_record_url": "https://ethics.ai/record/6618",
  "original_url": "https://arxiv.org/abs/2603.27693v1",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}