{
  "id": 565,
  "url": "https://arxiv.org/abs/2606.26918v1",
  "title": "Diagnosing Task Insensitivity in Language Agents",
  "summary": "Large language models can serve as capable long-horizon agents, but their out-of-distribution (OOD) generalization remains weak. We identify a key source of this failure as task insensitivity: when faced with similar but distinct tasks, models might apply patterns learned during training and fail to solve the task at hand. We show that models often continue with actions aligned with the original task even when the instruction is semantically corrupted and cannot be directly answered. We further ",
  "authors": "Jingyu Liu, Xiaopeng Wu, Kehan Chen, Chuan Yu, Yong Liu",
  "category": "research",
  "topics": "healthcare,agents-autonomy",
  "orgs": null,
  "regions": null,
  "published_at": "2026-06-25T11:53:41.000Z",
  "fetched_at": "2026-07-14T14:14:37.248Z",
  "source_slug": "arxiv-ethics",
  "source_name": "arXiv",
  "source_homepage": "https://arxiv.org",
  "ethics_ai_record_url": "https://ethics.ai/record/565",
  "original_url": "https://arxiv.org/abs/2606.26918v1",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}