{
  "id": 5829,
  "url": "https://arxiv.org/abs/2604.13803v1",
  "title": "Gaslight, Gatekeep, V1-V3: Early Visual Cortex Alignment Shields Vision-Language Models from Sycophantic Manipulation",
  "summary": "Vision-language models are increasingly deployed in high-stakes settings, yet their susceptibility to sycophantic manipulation remains poorly understood, particularly in relation to how these models represent visual information internally. Whether models whose visual representations more closely mirror human neural processing are also more resistant to adversarial pressure is an open question with implications for both neuroscience and AI safety. We investigate this question by evaluating 12 ope",
  "authors": "Arya Shah, Vaibhav Tripathi, Mayank Singh, Chaklam Silpasuwanchai",
  "category": "research",
  "topics": "safety-alignment,finance-investment",
  "orgs": null,
  "regions": null,
  "published_at": "2026-04-15T12:38:51.000Z",
  "fetched_at": "2026-07-14T16:32:02.061Z",
  "source_slug": "arxiv-ethics",
  "source_name": "arXiv",
  "source_homepage": "https://arxiv.org",
  "ethics_ai_record_url": "https://ethics.ai/record/5829",
  "original_url": "https://arxiv.org/abs/2604.13803v1",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}