{
  "id": 18369,
  "url": "https://arxiv.org/abs/2608.11171",
  "title": "From Interpretability to Control: Insights from Six Years of the TrustNLP Workshop",
  "summary": "arXiv:2608.11171v1 Announce Type: cross Abstract: The Workshop on Trustworthy Natural Language Processing (TrustNLP), co-located with major ACL conferences since 2021, has grown from 8 proceedings papers to 41 over six editions, documenting a field-wide transition from post-hoc interpretability of static models to mechanistic understanding and proactive control of generative systems. We synthesize insights from all 144 proceedings papers, classifying them along six trust dimensions grounded in e",
  "authors": "Rahul Gupta, Abhinav Mohanty, Anaelia Ovalle, Anil Ramakrishna, Anubrata Das, Apurv Verma, Jwala Dhamala, Ninareh Mehrabi, Tharindu Kumarage, Yada Pruksachatkun, Yang Trista Cao, Kai-Wei Chang, Aram Galstyan",
  "category": "research",
  "topics": "safety-alignment",
  "orgs": null,
  "regions": null,
  "published_at": "2026-08-12T04:00:00.000Z",
  "fetched_at": "2026-08-12T05:10:43.828Z",
  "source_slug": "arxiv-cscy",
  "source_name": "arXiv cs.CY",
  "source_homepage": "https://arxiv.org/list/cs.CY/recent",
  "ethics_ai_record_url": "https://ethics.ai/record/18369",
  "original_url": "https://arxiv.org/abs/2608.11171",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}