{
  "id": 5320,
  "url": "https://arxiv.org/abs/2604.24178v1",
  "title": "Meta-Aligner: Bidirectional Preference-Policy Optimization for Multi-Objective LLMs Alignment",
  "summary": "Multi-Objective Alignment aims to align Large Language Models (LLMs) with diverse and often conflicting human values by optimizing multiple objectives simultaneously. Existing methods predominantly rely on static preference weight construction strategies. However, rigidly aligning to fixed targets discards valuable intermediate information, as training responses inherently embody valid preference trade-offs even when deviating from the target. To address this limitation, we propose Meal, i.e., M",
  "authors": "Wenzhe Xu, Biao Liu, Yiyang Sun, Xin Geng, Ning Xu",
  "category": "research",
  "topics": "regulation,safety-alignment",
  "orgs": "meta",
  "regions": null,
  "published_at": "2026-04-27T08:36:13.000Z",
  "fetched_at": "2026-07-14T16:31:40.220Z",
  "source_slug": "arxiv-ethics",
  "source_name": "arXiv",
  "source_homepage": "https://arxiv.org",
  "ethics_ai_record_url": "https://ethics.ai/record/5320",
  "original_url": "https://arxiv.org/abs/2604.24178v1",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}