{
  "id": 5859,
  "url": "https://arxiv.org/abs/2604.13175v1",
  "title": "Pareto-Optimal Offline Reinforcement Learning via Smooth Tchebysheff Scalarization",
  "summary": "Large language models can be aligned with human preferences through offline reinforcement learning (RL) on small labeled datasets. While single-objective alignment is well-studied, many real-world applications demand the simultaneous optimization of multiple conflicting rewards, e.g. optimizing both catalytic activity and specificity in protein engineering, or helpfulness and harmlessness for chatbots. Prior work has largely relied on linear reward scalarization, but this approach provably fails",
  "authors": "Aadyot Bhatnagar, Peter Mørch Groth, Ali Madani",
  "category": "research",
  "topics": "safety-alignment,biotech",
  "orgs": null,
  "regions": null,
  "published_at": "2026-04-14T18:00:39.000Z",
  "fetched_at": "2026-07-14T16:32:02.062Z",
  "source_slug": "arxiv-ethics",
  "source_name": "arXiv",
  "source_homepage": "https://arxiv.org",
  "ethics_ai_record_url": "https://ethics.ai/record/5859",
  "original_url": "https://arxiv.org/abs/2604.13175v1",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}