{
  "id": 11340,
  "url": "https://arxiv.org/abs/2607.14952",
  "title": "LongStraw: Long-Context RL Beyond 2M Tokens under a Fixed GPU Budget",
  "summary": "A growing gap separates inference context lengths from RL post-training: inference systems are approaching million-token contexts, while post-training workloads often remain at 256K tokens or below and rely on length generalization at deployment. The gap is especially important for AI agents, whose observations, tool outputs, documents, and prior decisions accumulate over long trajectories. LongStraw is an architecture-aware execution stack for million-token RL post-training under a fixed GPU bu",
  "authors": "Changhai Zhou, Kieran Liu, Yuhua Zhou, Qian Qiao, Jun Gao, Harry Zhang",
  "category": "research",
  "topics": "agents-autonomy",
  "orgs": null,
  "regions": null,
  "published_at": "2026-07-15T20:00:00.000Z",
  "fetched_at": "2026-07-18T05:10:55.931Z",
  "source_slug": "hf-daily",
  "source_name": "HuggingFace Daily Papers",
  "source_homepage": "https://huggingface.co/papers",
  "ethics_ai_record_url": "https://ethics.ai/record/11340",
  "original_url": "https://arxiv.org/abs/2607.14952",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}