{
  "id": 5479,
  "url": "https://arxiv.org/abs/2604.21102v1",
  "title": "Leveraging Multimodal LLMs for Built Environment and Housing Attribute Assessment from Street-View Imagery",
  "summary": "We present a novel framework for automatically evaluating building conditions nationwide in the United States by leveraging large language models (LLMs) and Google Street View (GSV) imagery. By fine-tuning Gemma 3 27B on a modest human-labeled dataset, our approach achieves strong alignment with human mean opinion scores (MOS), outperforming even individual raters on SRCC and PLCC relative to the MOS benchmark. To enhance efficiency, we apply knowledge distillation, transferring the capabilities",
  "authors": "Siyuan Yao, Siavash Ghorbany, Kuangshi Ai, Arnav Cherukuthota, Meghan Forstchen, Alexis Korotasz et al.",
  "category": "research",
  "topics": "safety-alignment,environment",
  "orgs": "google",
  "regions": "us",
  "published_at": "2026-04-22T21:42:09.000Z",
  "fetched_at": "2026-07-14T16:31:48.872Z",
  "source_slug": "arxiv-ethics",
  "source_name": "arXiv",
  "source_homepage": "https://arxiv.org",
  "ethics_ai_record_url": "https://ethics.ai/record/5479",
  "original_url": "https://arxiv.org/abs/2604.21102v1",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}