{
  "id": 16527,
  "url": "https://arxiv.org/abs/2608.04006v1",
  "title": "Calibrating Trustworthiness: Co-Designing Metrics and Visualizations for Evaluating LLMs in Education",
  "summary": "LLMs are reshaping educational technology, yet evaluating their responses for pedagogical alignment remains underexplored, relying heavily on the expertise of learning engineers building the technology. To bridge this gap, we explore trustworthiness as a structured lens for evaluation, leveraging existing measures of LLM trustworthiness to systematically identify potential pedagogical disruptions. Through a longitudinal co-design process with learning engineers developing an LLM-powered digital",
  "authors": "Adam Coscia, Sujata Duwal, Langdon Holmes, Scott Crossley, Alex Endert",
  "category": "research",
  "topics": "safety-alignment,children-education",
  "orgs": null,
  "regions": null,
  "published_at": "2026-08-04T17:59:20.000Z",
  "fetched_at": "2026-08-05T05:10:44.550Z",
  "source_slug": "x-arxiv-cs-hc",
  "source_name": "arXiv cs.HC",
  "source_homepage": "https://arxiv.org/list/cs.HC/recent",
  "ethics_ai_record_url": "https://ethics.ai/record/16527",
  "original_url": "https://arxiv.org/abs/2608.04006v1",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}