{
  "id": 1017,
  "url": "https://arxiv.org/abs/2606.15507v1",
  "title": "Frame-Conditioned Moral Computation in LLaMA 3.1-8B-Instruct: A Mechanistic Interpretability Audit of Ethical Reasoning",
  "summary": "Behavioral audits of Large Language Models on moral prompts measure what the model says, not the internal computation producing it. We use Transluce, an AI-driven mechanistic-interpretability platform, to examine LLaMA 3.1-8B-Instruct on 54 moral prompts in four batteries: 17 dilemmas, policy, and meta-ethical questions (B1); 6 role-playing scenarios (B3); and a controlled trolley contrast varying the switching mechanism with people fixed (B4, 15 prompts) or identity attributes with mechanism fi",
  "authors": "Ali Dasdan, Manan Shah, W. Russell Neuman, Chad Coleman, Kund Meghani, Safinah Ali",
  "category": "research",
  "topics": "regulation,safety-alignment,transparency",
  "orgs": null,
  "regions": null,
  "published_at": "2026-06-13T23:24:36.000Z",
  "fetched_at": "2026-07-14T14:14:59.013Z",
  "source_slug": "arxiv-ethics",
  "source_name": "arXiv",
  "source_homepage": "https://arxiv.org",
  "ethics_ai_record_url": "https://ethics.ai/record/1017",
  "original_url": "https://arxiv.org/abs/2606.15507v1",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}