{
  "id": 4075,
  "url": "https://arxiv.org/abs/2605.19156v1",
  "title": "How Far Are We From True Auto-Research?",
  "summary": "Recent auto-research systems can produce complete papers, but feasibility is not the same as quality, and the field still lacks a systematic study of how good agent-generated papers actually are. We introduce ResearchArena, a minimal scaffold that lets off-the-shelf agents (Claude Code using Opus 4.6, Codex using GPT-5.4, and Kimi Code using K2.5) carry out the full research loop themselves (ideation, experimentation, paper writing, self-refinement) under only lightweight guidance. Across 13 com",
  "authors": "Zhengxin Zhang, Ning Wang, Sainyam Galhotra, Claire Cardie",
  "category": "research",
  "topics": "agents-autonomy",
  "orgs": "openai",
  "regions": null,
  "published_at": "2026-05-18T22:20:33.000Z",
  "fetched_at": "2026-07-14T16:30:45.936Z",
  "source_slug": "arxiv-ethics",
  "source_name": "arXiv",
  "source_homepage": "https://arxiv.org",
  "ethics_ai_record_url": "https://ethics.ai/record/4075",
  "original_url": "https://arxiv.org/abs/2605.19156v1",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}