{
  "id": 5955,
  "url": "https://arxiv.org/abs/2604.16487v2",
  "title": "Geometry-Aware CLIP Retrieval via Local Cross-Modal Alignment and Steering",
  "summary": "CLIP retrieval is typically framed as a pointwise similarity problem in a shared embedding space. While CLIP achieves strong global cross-modal alignment, many retrieval failures arise from local geometric inconsistencies: nearby items are incorrectly ordered, leading to systematic confusions (e.g., pentagon vs. hexagon) and produces diffuse, weakly controlled result sets. Prior work largely optimizes for point wise relevance or finetuning to mitigate these problems. We instead view retrieval as",
  "authors": "Nirmalendu Prakash, Narmeen Fatimah Oozeer, Xin Su, Phillip Howard, Shaan Shah, Zoe Wanying He et al.",
  "category": "research",
  "topics": "safety-alignment",
  "orgs": null,
  "regions": null,
  "published_at": "2026-04-13T08:27:11.000Z",
  "fetched_at": "2026-07-14T16:32:06.471Z",
  "source_slug": "arxiv-ethics",
  "source_name": "arXiv",
  "source_homepage": "https://arxiv.org",
  "ethics_ai_record_url": "https://ethics.ai/record/5955",
  "original_url": "https://arxiv.org/abs/2604.16487v2",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}