{
  "id": 4531,
  "url": "https://arxiv.org/abs/2605.11217v1",
  "title": "Leveraging RAG for Training-Free Alignment of LLMs",
  "summary": "Large language model (LLM) alignment algorithms typically consist of post-training over preference pairs. While such algorithms are widely used to enable safety guardrails and align LLMs with general human preferences, we show that state-of-the-art alignment algorithms require significant computational resources while being far less capable of enabling refusal guardrails for recent agentic attacks. Thus, to improve refusal guardrails against such attacks without drastically increasing computatio",
  "authors": "John T. Halloran",
  "category": "research",
  "topics": "safety-alignment,agents-autonomy",
  "orgs": null,
  "regions": null,
  "published_at": "2026-05-11T20:29:39.000Z",
  "fetched_at": "2026-07-14T16:31:03.580Z",
  "source_slug": "arxiv-ethics",
  "source_name": "arXiv",
  "source_homepage": "https://arxiv.org",
  "ethics_ai_record_url": "https://ethics.ai/record/4531",
  "original_url": "https://arxiv.org/abs/2605.11217v1",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}