{
  "id": 4871,
  "url": "https://arxiv.org/abs/2605.05611v2",
  "title": "X-Voice: Enabling Everyone to Speak 30 Languages via Zero-Shot Cross-Lingual Voice Cloning",
  "summary": "In this paper, we present X-Voice, a 0.4B multilingual zero-shot voice cloning model that clones arbitrary voices and enables everyone to speak 30 languages. X-Voice is trained on a 420K-hour multilingual corpus using the International Phonetic Alphabet (IPA) as a unified representation. To eliminate the reliance on prompt text without complex preprocessing like forced alignment, we design a two-stage training paradigm. In Stage 1, we establish X-Voice$_{\\text{s1}}$ through standard conditional ",
  "authors": "Rixi Xu, Qingyu Liu, Haitao Li, Yushen Chen, Zhikang Niu, Yunting Yang et al.",
  "category": "research",
  "topics": "safety-alignment",
  "orgs": "google",
  "regions": null,
  "published_at": "2026-05-07T02:57:53.000Z",
  "fetched_at": "2026-07-14T16:31:21.930Z",
  "source_slug": "arxiv-ethics",
  "source_name": "arXiv",
  "source_homepage": "https://arxiv.org",
  "ethics_ai_record_url": "https://ethics.ai/record/4871",
  "original_url": "https://arxiv.org/abs/2605.05611v2",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}