{
  "id": 5844,
  "url": "https://arxiv.org/abs/2604.13517v4",
  "title": "Representation over Routing: Diagnosing Temporal Routing Pathologies in Multi-Timescale PPO",
  "summary": "Temporal credit assignment in reinforcement learning is often approached by introducing value estimates at multiple discount factors. A natural next step is to let the actor dynamically route among these temporal heads, using either differentiable attention or heuristic uncertainty weights. This paper argues that such routing can create a numerical shortcut rather than a reliable temporal abstraction. We study this issue in a controlled PPO setting on LunarLander-v2, using the environment as a v",
  "authors": "Jing Sun",
  "category": "research",
  "topics": "healthcare,environment",
  "orgs": null,
  "regions": null,
  "published_at": "2026-04-15T06:03:07.000Z",
  "fetched_at": "2026-07-14T16:32:02.061Z",
  "source_slug": "arxiv-ethics",
  "source_name": "arXiv",
  "source_homepage": "https://arxiv.org",
  "ethics_ai_record_url": "https://ethics.ai/record/5844",
  "original_url": "https://arxiv.org/abs/2604.13517v4",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}