{
  "id": 4571,
  "url": "https://arxiv.org/abs/2605.10261v1",
  "title": "E-TCAV: Formalizing Penultimate Proxies for Efficient Concept Based Interpretability",
  "summary": "TCAV (Testing with Concept Activation Vectors) is an interpretability method that assesses the alignment between the internal representations of a trained neural network and human-understandable, high-level concepts. Though effective, TCAV suffers from significant computational overhead, inter-layer disagreement of TCAV scores, and statistical instability. This work takes a step toward addressing these challenges by introducing E-TCAV, a framework for efficient approximation of TCAV scores, whic",
  "authors": "Hasib Aslam, Muhammad Ali Chattha, Muhammad Taha Mukhtar, Muhammad Imran Malik, Andreas Dengel, Sheraz Ahmed",
  "category": "research",
  "topics": "safety-alignment",
  "orgs": null,
  "regions": null,
  "published_at": "2026-05-11T09:25:41.000Z",
  "fetched_at": "2026-07-14T16:31:08.353Z",
  "source_slug": "arxiv-ethics",
  "source_name": "arXiv",
  "source_homepage": "https://arxiv.org",
  "ethics_ai_record_url": "https://ethics.ai/record/4571",
  "original_url": "https://arxiv.org/abs/2605.10261v1",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}