{
  "id": 14858,
  "url": "https://arxiv.org/abs/2607.26094v1",
  "title": "Meta-Learned Reward Shaping for Reinforcement Learning from Human Feedback",
  "summary": "Reinforcement Learning from Human Feedback (RLHF) is the standard approach for aligning large language models with human preferences, but its quality is limited by static, task-agnostic reward models. This mismatch leads to sparse learning signals and suboptimal alignment. We introduce MeRLa (Meta-Learned Reward Shaping), a principled framework that meta-learns a task-aware shaping function $Φ(x,y;φ)$ across auxiliary tasks before RLHF training. The learned shaping produces a composite reward th",
  "authors": "Yunpeng Chu",
  "category": "research",
  "topics": "safety-alignment",
  "orgs": "meta",
  "regions": null,
  "published_at": "2026-07-28T02:09:26.000Z",
  "fetched_at": "2026-07-30T05:10:24.387Z",
  "source_slug": "arxiv-cslg",
  "source_name": "arXiv cs.LG",
  "source_homepage": "https://arxiv.org/list/cs.LG/recent",
  "ethics_ai_record_url": "https://ethics.ai/record/14858",
  "original_url": "https://arxiv.org/abs/2607.26094v1",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}