{
  "id": 1192,
  "url": "https://arxiv.org/abs/2606.10912v1",
  "title": "What Do Deepfake Speech Detectors Actually Hear?",
  "summary": "Deepfake speech detectors often output a single score without explaining why an audio sample is flagged, where in the signal the evidence lies, or what cues drive the decision. We propose an audio-native explainability pipeline using Integrated Gradients on time-aligned self-supervised representations to localize decision evidence over time. We apply the proposed method to three WavLM-based detectors (AASIST, CA-MHFA, SLS) on ASVspoof 5 and manually annotate the highest-attribution regions to pr",
  "authors": "Vojtěch Staněk, Veronika Jirmusová, Anton Firc, Kamil Malinka, Jakub Reš, Martin Perešíni",
  "category": "research",
  "topics": "misinformation,transparency",
  "orgs": null,
  "regions": null,
  "published_at": "2026-06-09T14:21:45.000Z",
  "fetched_at": "2026-07-14T14:15:03.617Z",
  "source_slug": "arxiv-ethics",
  "source_name": "arXiv",
  "source_homepage": "https://arxiv.org",
  "ethics_ai_record_url": "https://ethics.ai/record/1192",
  "original_url": "https://arxiv.org/abs/2606.10912v1",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}