{
  "id": 3140,
  "url": "https://arxiv.org/abs/2607.05462v1",
  "title": "Evaluating calibrated refusal and safe usefulness in dual-use biology settings",
  "summary": "As AI agents are incorporated into life science workflows, the capabilities that speed discovery might also enable misuse. We present BioSecBench-Refusal, a benchmark for risk identification and refusal behavior for biological research tasks. The benchmark pairs 61 Routine tasks, legitimate analyses adapted from the published literature, with 46 Red-Team tasks, fictional scenarios that resemble real research but conceal a biosecurity hazard. Across 16 model-harness configurations, refusal rates ",
  "authors": "Edwin H. Wintermute, Harmon Bhasin, Christina M. Agapakis, Dianzhuo Wang, Evan Seeyave, Arjun Banerjee, Daniel Fulop, Matthew C. Watson, Adam J. Meyer, Sandrine Boissel, Jens H. Kuhn, Rishi Jain, Noah D. Taylor, Helena Shomar, Patrick M. Boyle, Kenny Workman",
  "category": "research",
  "topics": "safety-alignment,agents-autonomy,biotech",
  "orgs": null,
  "regions": null,
  "published_at": "2026-07-06T01:46:07.000Z",
  "fetched_at": "2026-07-14T16:11:46.979Z",
  "source_slug": "x-arxiv-red-teaming-query",
  "source_name": "arXiv red teaming query",
  "source_homepage": "https://arxiv.org/a/redteam",
  "ethics_ai_record_url": "https://ethics.ai/record/3140",
  "original_url": "https://arxiv.org/abs/2607.05462v1",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}