{
  "id": 1364,
  "url": "https://arxiv.org/abs/2606.07309v1",
  "title": "Acoustic Cue Alignment in Audio Language Models for Speech Emotion Recognition",
  "summary": "Instruction-following audio language models (ALMs) can be augmented with explicit acoustic cues, yet it remains unclear whether such cues are used in a grounded way when the raw audio is already available. We study this question in speech emotion recognition (SER) by deriving six interpretable acoustic concept tokens from the standardised eGeMAPS paralinguistic feature set. These tokens summarise energy, pitch, dynamics, brightness, formants, and voice quality, and are appended to the textual pr",
  "authors": "Iosif Tsangko, Andreas Triantafyllopoulos, Björn W. Schuller",
  "category": "research",
  "topics": "safety-alignment,environment",
  "orgs": null,
  "regions": null,
  "published_at": "2026-06-05T14:26:06.000Z",
  "fetched_at": "2026-07-14T14:15:12.458Z",
  "source_slug": "arxiv-ethics",
  "source_name": "arXiv",
  "source_homepage": "https://arxiv.org",
  "ethics_ai_record_url": "https://ethics.ai/record/1364",
  "original_url": "https://arxiv.org/abs/2606.07309v1",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}