{
  "id": 6556,
  "url": "https://arxiv.org/abs/2603.29467v1",
  "title": "M-MiniGPT4: Multilingual VLLM Alignment via Translated Data",
  "summary": "This paper presents a Multilingual Vision Large Language Model, named M-MiniGPT4. Our model exhibits strong vision-language understanding (VLU) capabilities across 11 languages. We utilize a mixture of native multilingual and translated data to push the multilingual VLU performance of the MiniGPT4 architecture. In addition, we propose a multilingual alignment training stage that uses parallel text corpora to further enhance the multilingual capabilities of our model. M-MiniGPT4 achieves 36% accu",
  "authors": "Seung Hun Han, Youssef Mohamed, Mohamed Elhoseiny",
  "category": "research",
  "topics": "safety-alignment",
  "orgs": null,
  "regions": null,
  "published_at": "2026-03-31T09:13:38.000Z",
  "fetched_at": "2026-07-14T16:32:33.103Z",
  "source_slug": "arxiv-ethics",
  "source_name": "arXiv",
  "source_homepage": "https://arxiv.org",
  "ethics_ai_record_url": "https://ethics.ai/record/6556",
  "original_url": "https://arxiv.org/abs/2603.29467v1",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}