{
  "id": 792,
  "url": "https://arxiv.org/abs/2606.20508v1",
  "title": "What Do Safety-Aligned LLMs Learn From Mixed Compliance Demonstrations?",
  "summary": "Prior work has shown that in-context demonstrations can jailbreak language models, but it remains unclear how models interpret different types of compliance demonstrations. We study this by mixing benign compliance demonstrations (non-harmful request, helpful response) with harmful compliance demonstrations (harmful request, helpful response) and testing three hypotheses about how demonstration composition drives harmful compliance. Across four models, we find that benign and harmful demonstrati",
  "authors": "Sihui Dai, Mann Patel",
  "category": "research",
  "topics": "regulation,safety-alignment",
  "orgs": null,
  "regions": null,
  "published_at": "2026-06-18T17:25:38.000Z",
  "fetched_at": "2026-07-14T14:14:46.036Z",
  "source_slug": "arxiv-ethics",
  "source_name": "arXiv",
  "source_homepage": "https://arxiv.org",
  "ethics_ai_record_url": "https://ethics.ai/record/792",
  "original_url": "https://arxiv.org/abs/2606.20508v1",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}