{
  "id": 11861,
  "url": "https://arxiv.org/abs/2607.15755v1",
  "title": "AuEmoChat: Authentic Emotion Understanding and Rendering for Conversational Speech Synthesis",
  "summary": "Conversational Speech Synthesis (CSS) aims to synthesize speech with human-like emotional expression and contextual consistency in user-agent interactions. Existing CSS methods struggle to render authentic human emotions due to limited predefined emotion label spaces (e.g., seven emotion categories), while redundant multimodal tokens in multi-turn dialogue history interfere with context understanding. To address these issues, we propose AuEmoChat, a CSS framework for authentic emotion understand",
  "authors": "Zhenqi Jia, Yuan Zhao, Aruukhan, Rui Liu, Haizhou Li",
  "category": "research",
  "topics": "agents-autonomy",
  "orgs": null,
  "regions": null,
  "published_at": "2026-07-17T08:47:58.000Z",
  "fetched_at": "2026-07-20T05:10:09.534Z",
  "source_slug": "x-arxiv-cs-ai",
  "source_name": "arXiv cs.AI",
  "source_homepage": "https://arxiv.org/list/cs.AI/recent",
  "ethics_ai_record_url": "https://ethics.ai/record/11861",
  "original_url": "https://arxiv.org/abs/2607.15755v1",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}