{
  "id": 7481,
  "url": "https://arxiv.org/abs/2603.08483v1",
  "title": "X-AVDT: Audio-Visual Cross-Attention for Robust Deepfake Detection",
  "summary": "The surge of highly realistic synthetic videos produced by contemporary generative systems has significantly increased the risk of malicious use, challenging both humans and existing detectors. Against this backdrop, we take a generator-side view and observe that internal cross-attention mechanisms in these models encode fine-grained speech-motion alignment, offering useful correspondence cues for forgery detection. Building on this insight, we propose X-AVDT, a robust and generalizable deepfake",
  "authors": "Youngseo Kim, Kwan Yun, Seokhyeon Hong, Sihun Cha, Colette Suhjung Koo, Junyong Noh",
  "category": "research",
  "topics": "safety-alignment,misinformation",
  "orgs": null,
  "regions": null,
  "published_at": "2026-03-09T15:18:42.000Z",
  "fetched_at": "2026-07-14T16:33:16.668Z",
  "source_slug": "arxiv-ethics",
  "source_name": "arXiv",
  "source_homepage": "https://arxiv.org",
  "ethics_ai_record_url": "https://ethics.ai/record/7481",
  "original_url": "https://arxiv.org/abs/2603.08483v1",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}