{
  "id": 3952,
  "url": "https://arxiv.org/abs/2605.21683v1",
  "title": "Investigating Concept Alignment Using Implausible Category Members",
  "summary": "Developing AI systems with a human-like understanding of everyday concepts is a key step towards developing safe, reliable systems whose behavior makes sense to humans. When probing concept understanding, asking questions about plausible category members (e.g., \"Is a car a vehicle?\") is likely to recall patterns in the model's vast training data. We pursue an alternative strategy, characterizing the boundaries of conceptual categories by asking about implausible category members (e.g., \"Is an ol",
  "authors": "Sunayana Rane, Brenden M. Lake, Thomas L. Griffiths",
  "category": "research",
  "topics": "safety-alignment,finance-investment",
  "orgs": null,
  "regions": null,
  "published_at": "2026-05-20T19:41:35.000Z",
  "fetched_at": "2026-07-14T16:30:36.744Z",
  "source_slug": "arxiv-ethics",
  "source_name": "arXiv",
  "source_homepage": "https://arxiv.org",
  "ethics_ai_record_url": "https://ethics.ai/record/3952",
  "original_url": "https://arxiv.org/abs/2605.21683v1",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}