{
  "id": 4234,
  "url": "https://arxiv.org/abs/2605.18882v1",
  "title": "To Call or Not to Call: Diagnosing Intrinsic Over-Calling Bias in LLM Agents",
  "summary": "LLM agents exhibit a consistent tendency to over-call, invoking tools even in situations where none is needed. On the When2Call benchmark, six models from three families show high call accuracy but much lower no-call accuracy, leaving overall accuracy in the 55%-70% range. We trace this to an Intrinsic Bias Hypothesis (IBH): the call/no-call decision mapping carries an activation-independent call offset, so the model favors call even at activation parity. Using Sparse Autoencoders (SAEs), we rec",
  "authors": "Wei Shi, Ziheng Peng, Sihang Li, Xiting Wang, Xiang Wang, Mengnan Du et al.",
  "category": "research",
  "topics": "bias-fairness,healthcare,agents-autonomy",
  "orgs": null,
  "regions": null,
  "published_at": "2026-05-16T04:18:30.000Z",
  "fetched_at": "2026-07-14T16:30:50.573Z",
  "source_slug": "arxiv-ethics",
  "source_name": "arXiv",
  "source_homepage": "https://arxiv.org",
  "ethics_ai_record_url": "https://ethics.ai/record/4234",
  "original_url": "https://arxiv.org/abs/2605.18882v1",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}