{
  "id": 18380,
  "url": "https://arxiv.org/abs/2608.10366",
  "title": "DSAgentBench: Can Agents Automate End-to-End Data-Science Workflows in Real Computer Environments?",
  "summary": "Real-world data science involves long-horizon workflows that span data wrangling, exploration, modeling, visualization, and validation, and require coordinated use of tools such as notebooks, IDEs, terminals, browsers, and databases within real operating environments. Yet existing benchmarks lack real-computer interaction and do not evaluate whether agents can execute complete end-to-end data-science workflows in realistic computing environments, failing to capture the multi-stage, multi-tool na",
  "authors": "Mizanur Rahman, Mohammed Saidul Islam, Ridwan Mahbub, Md Tahmid Rahman Laskar, Shafiq Joty, Enamul Hoque Prince",
  "category": "research",
  "topics": "agents-autonomy,environment",
  "orgs": null,
  "regions": null,
  "published_at": "2026-08-10T20:00:00.000Z",
  "fetched_at": "2026-08-12T05:10:43.828Z",
  "source_slug": "hf-daily",
  "source_name": "HuggingFace Daily Papers",
  "source_homepage": "https://huggingface.co/papers",
  "ethics_ai_record_url": "https://ethics.ai/record/18380",
  "original_url": "https://arxiv.org/abs/2608.10366",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}