{
  "id": 5840,
  "url": "https://arxiv.org/abs/2604.13561v1",
  "title": "CLIP Architecture for Abdominal CT Image-Text Alignment and Zero-Shot Learning: Investigating Batch Composition and Data Scaling",
  "summary": "Vision-language models trained with contrastive learning on paired medical images and reports show strong zero-shot diagnostic capabilities, yet the effect of training batch composition on learned representations remains unexplored for 3D medical imaging. We reproduce Merlin, a dual-encoder model that aligns 3D abdominal CT volumes with radiology reports using symmetric InfoNCE loss, achieving a zero-shot macro F1 of 74.45% across 30 findings (original: 73.00%). We then investigate two axes of v",
  "authors": "Shivika, Kartik Bose, Pankaj Gupta",
  "category": "research",
  "topics": "safety-alignment,healthcare,finance-investment",
  "orgs": null,
  "regions": null,
  "published_at": "2026-04-15T07:10:01.000Z",
  "fetched_at": "2026-07-14T16:32:02.061Z",
  "source_slug": "arxiv-ethics",
  "source_name": "arXiv",
  "source_homepage": "https://arxiv.org",
  "ethics_ai_record_url": "https://ethics.ai/record/5840",
  "original_url": "https://arxiv.org/abs/2604.13561v1",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}