{
  "id": 5275,
  "url": "https://arxiv.org/abs/2604.25249v1",
  "title": "Below-Chance Blindness: Prompted Underperformance in Small LLMs Produces Positional Bias Rather than Answer Avoidance",
  "summary": "Detecting sandbagging--the deliberate underperformance on capability evaluations--is an open problem in AI safety. We tested whether symptom validity testing (SVT) logic from clinical malingering detection could identify sandbagging through below-chance performance (BCB) on forced-choice items. In a pre-registered pilot at the 7-9 billion parameter instruction-tuned scale (3 models, 4 MMLU-Pro domains, 4 conditions, 500 items per cell, 24,000 total trials), the plausibility gate failed. Zero of ",
  "authors": "Jon-Paul Cacioli",
  "category": "research",
  "topics": "bias-fairness,safety-alignment,healthcare",
  "orgs": null,
  "regions": null,
  "published_at": "2026-04-28T05:57:23.000Z",
  "fetched_at": "2026-07-14T16:31:40.218Z",
  "source_slug": "arxiv-ethics",
  "source_name": "arXiv",
  "source_homepage": "https://arxiv.org",
  "ethics_ai_record_url": "https://ethics.ai/record/5275",
  "original_url": "https://arxiv.org/abs/2604.25249v1",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}