{
  "id": 4110,
  "url": "https://arxiv.org/abs/2605.18529v1",
  "title": "AMR-SD: Asymmetric Meta-Reflective Self-Distillation for Token-Level Credit Assignment",
  "summary": "The alignment of Large Language Models (LLMs) for complex reasoning heavily relies on Reinforcement Learning with Verifiable Rewards (RLVR). However, standard algorithms like GRPO apply sequence-level rewards uniformly to all tokens, creating a severe credit-assignment bottleneck. While on-policy self-distillation attempts to resolve this by conditioning a self-teacher on privileged contexts, direct exposure to raw oracle solutions often induces over-conditioned teacher distributions, implicit a",
  "authors": "Zhenlin Wei, Pu Jian, Yingzhuo Deng, Xiaohan Wang, Jiajun Chai, Zhexin Hu et al.",
  "category": "research",
  "topics": "regulation,safety-alignment",
  "orgs": "meta",
  "regions": null,
  "published_at": "2026-05-18T15:14:34.000Z",
  "fetched_at": "2026-07-14T16:30:45.939Z",
  "source_slug": "arxiv-ethics",
  "source_name": "arXiv",
  "source_homepage": "https://arxiv.org",
  "ethics_ai_record_url": "https://ethics.ai/record/4110",
  "original_url": "https://arxiv.org/abs/2605.18529v1",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}