{
  "id": 17809,
  "url": "https://arxiv.org/abs/2608.07068v1",
  "title": "MemOPD: On-Policy Distillation through Memory State Alignment for Long-Horizon Agents",
  "summary": "Long-horizon agents accumulate growing contexts during interaction, impairing performance and stability. Compact memory mitigates this problem by compressing and rewriting the history retained between model invocations. Learning what to retain typically relies on proximal policy optimization (PPO) with final task rewards, but sparse rewards provide little guidance for individual memory updates. This limitation motivates on-policy distillation (OPD), which supplies dense teacher supervision on st",
  "authors": "Zhiyuan Liu, Tinghong Ye, Chenghao Liu, Yizhuo Li, Songfang Huang",
  "category": "research",
  "topics": "regulation,safety-alignment,agents-autonomy",
  "orgs": null,
  "regions": null,
  "published_at": "2026-08-07T10:18:31.000Z",
  "fetched_at": "2026-08-10T05:10:00.488Z",
  "source_slug": "arxiv-ethics",
  "source_name": "arXiv",
  "source_homepage": "https://arxiv.org",
  "ethics_ai_record_url": "https://ethics.ai/record/17809",
  "original_url": "https://arxiv.org/abs/2608.07068v1",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}