{
  "id": 12593,
  "url": "https://arxiv.org/abs/2607.19262v1",
  "title": "BioSecBench-Surveillance: A Verifiable Benchmark for AI Agents in Pathogen Genomic Surveillance",
  "summary": "As pathogen genomic surveillance scales, the bottleneck is shifting from data generation to analysis. We present BioSecBench-Surveillance, a verifiable benchmark of 100 evaluations testing whether AI agents can infer the right analysis pipeline from raw sequencing data and surveillance context. Each evaluation gives an agent only the data and context a human analyst would have, then grades its structured answer deterministically. The tasks span seven categories, from taxonomic classification to",
  "authors": "Harmon Bhasin, Kevin Flyangolts, Dianzhuo Wang, Evan Seeyave, Arjun Banerjee, Amanda Darling, Joshua Stallings, David Stern, Shawn Higdon, Claire Duvallet, Bryan Tegomoh, Kenny Workman",
  "category": "research",
  "topics": "privacy-surveillance,agents-autonomy,biotech",
  "orgs": null,
  "regions": null,
  "published_at": "2026-07-21T16:33:57.000Z",
  "fetched_at": "2026-07-22T05:10:49.469Z",
  "source_slug": "x-arxiv-cs-ai",
  "source_name": "arXiv cs.AI",
  "source_homepage": "https://arxiv.org/list/cs.AI/recent",
  "ethics_ai_record_url": "https://ethics.ai/record/12593",
  "original_url": "https://arxiv.org/abs/2607.19262v1",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}