{
  "id": 12656,
  "url": "https://arxiv.org/abs/2607.20207",
  "title": "SeededGrasp: Language-Guided Grasping in Complex Scenes with Multiple Embodiments",
  "summary": "Practical robotic grasping in complex scenes requires both 3D spatial reasoning and alignment with task-specific requirements. Vision-language models (VLMs) offer a natural way to specify these requirements using language, but existing approaches either use a VLM to predict the grasp directly with limited spatial awareness, or train the VLM together with the grasping model, which requires significantly more data and compute. These limitations impede performance and have prevented scaling to mult",
  "authors": "Yang Xu, Gurpreet Singh Mukker, Raymond Wang, Jasper Gerigk, Maria Attarian, Igor Gilitschenski",
  "category": "research",
  "topics": "safety-alignment,agents-autonomy",
  "orgs": null,
  "regions": null,
  "published_at": "2026-07-21T20:00:00.000Z",
  "fetched_at": "2026-07-23T05:10:49.458Z",
  "source_slug": "hf-daily",
  "source_name": "HuggingFace Daily Papers",
  "source_homepage": "https://huggingface.co/papers",
  "ethics_ai_record_url": "https://ethics.ai/record/12656",
  "original_url": "https://arxiv.org/abs/2607.20207",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}