{
  "id": 12233,
  "url": "https://arxiv.org/abs/2607.17136v1",
  "title": "Teach it to stop, not just to click",
  "summary": "Agentic computer-use RL is reported in single runs, and those numbers mislead. Using verifier-guided repair of a 35B computer-use agent (CUA) across five oracle-graded environments, we show a repaired policy's success rate is dominated by upstream variance: a variance-components decomposition across three cells (crossed data-draw $\\times$ seed grid, bootstrap CIs) finds evaluation variance negligible ($σ_{\\mathrm{eval}} \\approx 0$) and the training-seed effect small everywhere ($\\leq 10\\%$); ins",
  "authors": "Barada Sahu, Shivesh Pandey",
  "category": "research",
  "topics": "regulation,agents-autonomy,environment",
  "orgs": null,
  "regions": null,
  "published_at": "2026-07-19T08:46:06.000Z",
  "fetched_at": "2026-07-21T05:10:12.656Z",
  "source_slug": "x-arxiv-cs-hc",
  "source_name": "arXiv cs.HC",
  "source_homepage": "https://arxiv.org/list/cs.HC/recent",
  "ethics_ai_record_url": "https://ethics.ai/record/12233",
  "original_url": "https://arxiv.org/abs/2607.17136v1",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}