{
  "id": 18020,
  "url": "https://arxiv.org/abs/2608.09270v1",
  "title": "GRASP: Granularity-Aware Region Alignment and Semantic Prototype Learning for Fine-Grained Cross-Modal Understanding in Drone Views",
  "summary": "Fine-grained cross-modal understanding in drone views is essential for aerial vision-language navigation. However, the inherent wide field of view and overhead perspective of drone scenarios impose dual challenges on vision-language understanding. At the macro level, overwhelming background clutter in visual representations leads to Cross-Modal Focus Misalignment, where the model prioritizes global environmental similarities over specific object details. At the micro level, Visual Isomorphism cr",
  "authors": "Jiahui Cui, Yan Zhao, Kan Wei, Enze Zhu, Peirong Zhang, Lei Wang et al.",
  "category": "research",
  "topics": "safety-alignment,environment",
  "orgs": null,
  "regions": null,
  "published_at": "2026-08-10T08:26:47.000Z",
  "fetched_at": "2026-08-11T05:10:37.351Z",
  "source_slug": "arxiv-ethics",
  "source_name": "arXiv",
  "source_homepage": "https://arxiv.org",
  "ethics_ai_record_url": "https://ethics.ai/record/18020",
  "original_url": "https://arxiv.org/abs/2608.09270v1",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}