{
  "id": 16147,
  "url": "https://arxiv.org/abs/2608.02446v1",
  "title": "Advancing Relevance Measurement with Vision-Language Models for Web-Scale Search",
  "summary": "Relevance evaluation plays a crucial role in personalized search systems, serving as a guardrail alongside user engagement metrics to ensure that search results align with user queries and intent. While human annotation is the traditional method for relevance evaluation, its high cost and long turnaround time limit its scalability. In this work, we present a VLM-based automated relevance evaluation pipeline deployed within Pinterest Search for online A/B experiments. We rigorously validate the a",
  "authors": "Han Wang, Alex Whitworth, Pak Ming Cheung, Zhenjie Zhang, Krishna Kamath, Xi Chen et al.",
  "category": "research",
  "topics": "safety-alignment",
  "orgs": null,
  "regions": null,
  "published_at": "2026-08-03T16:23:49.000Z",
  "fetched_at": "2026-08-04T05:10:21.797Z",
  "source_slug": "arxiv-cslg",
  "source_name": "arXiv cs.LG",
  "source_homepage": "https://arxiv.org/list/cs.LG/recent",
  "ethics_ai_record_url": "https://ethics.ai/record/16147",
  "original_url": "https://arxiv.org/abs/2608.02446v1",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}