{
  "id": 345,
  "url": "https://arxiv.org/abs/2607.01153v1",
  "title": "Adversarial Pragmatics for AI Safety Evaluation: A Benchmark for Instruction Conflict, Embedded Commands, and Policy Ambiguity",
  "summary": "Safety evaluations for language models increasingly depend on judgments about ambiguous natural-language behaviour: whether a model has followed an instruction, refused appropriately, complied with a policy, resisted an embedded command, or misreported progress in an agentic task. Existing benchmarks often compress these distinctions into pass/fail labels, obscuring whether failures arise from capability limits, policy ambiguity, instruction conflict, scaffold failure, or unstable evaluator judg",
  "authors": "Brett Reynolds",
  "category": "research",
  "topics": "regulation,safety-alignment,agents-autonomy",
  "orgs": null,
  "regions": null,
  "published_at": "2026-07-01T16:33:14.000Z",
  "fetched_at": "2026-07-14T14:14:28.436Z",
  "source_slug": "arxiv-ethics",
  "source_name": "arXiv",
  "source_homepage": "https://arxiv.org",
  "ethics_ai_record_url": "https://ethics.ai/record/345",
  "original_url": "https://arxiv.org/abs/2607.01153v1",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}