{
  "id": 13767,
  "url": "https://arxiv.org/abs/2607.22798",
  "title": "StateAct: Program State, before Pixels, for Long-Horizon Computer-Use Agents",
  "summary": "Computer-use agents are usually improved by strengthening perception: better models for reading a screenshot and choosing where to click. Yet a screenshot is only a lossy rendering of the underlying program state, e.g., the files, application backends, and DOM that hold the task data. Different states can produce the same pixels, while code can inspect and modify that state directly. StateAct is a code-first, multi-agent harness built around this distinction. Its main agent works directly with p",
  "authors": "Yan Yang, Xiangru Jian, Ziyang Luo, Zirui Zhao, Yutong Dai, Ziji Shi",
  "category": "research",
  "topics": "agents-autonomy",
  "orgs": null,
  "regions": null,
  "published_at": "2026-07-23T20:00:00.000Z",
  "fetched_at": "2026-07-28T05:10:12.325Z",
  "source_slug": "hf-daily",
  "source_name": "HuggingFace Daily Papers",
  "source_homepage": "https://huggingface.co/papers",
  "ethics_ai_record_url": "https://ethics.ai/record/13767",
  "original_url": "https://arxiv.org/abs/2607.22798",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}