{
  "id": 16539,
  "url": "https://arxiv.org/abs/2608.03979v1",
  "title": "Video-DeepResearch: Towards the Next-Generation Multimodal Deepresearch Agent",
  "summary": "We introduce Video-DeepResearch (Video-DR), extending multimodal agents from static images to continuous video streams, a setting that demands dense spatiotemporal grounding coupled with open-web exploration. Preliminary evaluations reveal two critical bottlenecks in current models: (1) modality bias, where agents bypass visual tools in favor of textual search, and (2) parametric knowledge leakage, where models rely on internal memory rather than genuine tool-augmented execution. To address thes",
  "authors": "Zhen Fang, Yu Zeng, Wenxuan Huang, Yiming Zhao, Shiting Huang, Tianfei Ren, Qi Lu, Qingnan Ren, Qisheng Su, Lionel Z. Wang, Qingyu Yin, Shuang Chen, Zehui Chen, Lin Chen, Zhenfei Yin, Yao Hu, Shaohui Lin, Wanli Ouyang, Shaosheng Cao, Feng Zhao",
  "category": "research",
  "topics": "bias-fairness,agents-autonomy",
  "orgs": null,
  "regions": null,
  "published_at": "2026-08-04T17:45:16.000Z",
  "fetched_at": "2026-08-05T05:10:44.550Z",
  "source_slug": "x-arxiv-cs-ai",
  "source_name": "arXiv cs.AI",
  "source_homepage": "https://arxiv.org/list/cs.AI/recent",
  "ethics_ai_record_url": "https://ethics.ai/record/16539",
  "original_url": "https://arxiv.org/abs/2608.03979v1",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}