{
  "id": 4732,
  "url": "https://arxiv.org/abs/2605.07699v1",
  "title": "DRIP-R: A Benchmark for Decision-Making and Reasoning Under Real-World Policy Ambiguity in the Retail Domain",
  "summary": "LLM-based agents are increasingly deployed for routine but consequential tasks in real-world domains, where their behavior is governed by inherently ambiguous domain policies that admit multiple valid interpretations. Despite the prevalence of such ambiguities in practice, existing agent benchmarks largely assume unambiguous, well-specified policies, leaving a critical evaluation gap. We introduce DRIP-R, a benchmark that systematically exploits real-world retail policy ambiguities to construct ",
  "authors": "Hsuvas Borkakoty, Sebastian Pohl, Cheng Wang, Bei Chen, Yufang Hou",
  "category": "research",
  "topics": "regulation,agents-autonomy",
  "orgs": null,
  "regions": null,
  "published_at": "2026-05-08T13:10:49.000Z",
  "fetched_at": "2026-07-14T16:31:12.746Z",
  "source_slug": "arxiv-ethics",
  "source_name": "arXiv",
  "source_homepage": "https://arxiv.org",
  "ethics_ai_record_url": "https://ethics.ai/record/4732",
  "original_url": "https://arxiv.org/abs/2605.07699v1",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}