{
  "id": 3211,
  "url": "https://arxiv.org/abs/2606.03626v1",
  "title": "TurtleAI: Benchmarking Multimodal Models for Visual Programming in Turtle Graphics",
  "summary": "Vision-language models (VLMs) have been explored for visual programming, where they generate code to solve visual tasks. However, most prior work focuses on visual programming for productivity; it remains unclear how well current VLMs perform on education-oriented visual programming and what factors limit their performance. To bridge this gap, we introduce TurtleAI, a benchmark containing 823 tasks curated based on real-world visual programming tasks in the Turtle Graphics domain. Solving these ",
  "authors": "Chao Wen, Jacqueline Staub, Adish Singla",
  "category": "research",
  "topics": "jobs-economy,children-education",
  "orgs": null,
  "regions": null,
  "published_at": "2026-06-02T13:25:05.000Z",
  "fetched_at": "2026-07-14T16:30:05.530Z",
  "source_slug": "arxiv-ethics",
  "source_name": "arXiv",
  "source_homepage": "https://arxiv.org",
  "ethics_ai_record_url": "https://ethics.ai/record/3211",
  "original_url": "https://arxiv.org/abs/2606.03626v1",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}