{
  "id": 5382,
  "url": "https://arxiv.org/abs/2604.23079v1",
  "title": "From Pixels to Explanations: Interpretable Diabetic Retinopathy Grading with CNN-Transformer Ensembles, Visual Explainability and Vision-Language Models",
  "summary": "The quality of diabetic retinopathy (DR) screening relies on the ability to correctly grade severity; however, many deep-learning (DL) classifiers cannot be easily interpreted in the clinical context. This study presents a methodology that combines strong discriminative models with multimodal explanations, converting retinal pixels into clinically interpretable outputs. Using the APTOS 2019 benchmark, we evaluated six representative CNN- and transformer-based backbones under a controlled protoco",
  "authors": "Pir Bakhsh Khokhar, Carmine Gravino, Fabio Palomba, Sule Yildirim Yayilgan, Sarang Shaikh",
  "category": "research",
  "topics": "bias-fairness,healthcare,transparency",
  "orgs": null,
  "regions": null,
  "published_at": "2026-04-25T00:21:11.000Z",
  "fetched_at": "2026-07-14T16:31:44.623Z",
  "source_slug": "arxiv-ethics",
  "source_name": "arXiv",
  "source_homepage": "https://arxiv.org",
  "ethics_ai_record_url": "https://ethics.ai/record/5382",
  "original_url": "https://arxiv.org/abs/2604.23079v1",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}