{
  "id": 5251,
  "url": "https://arxiv.org/abs/2604.25779v1",
  "title": "Sustained Gradient Alignment Mediates Subliminal Learning in a Multi-Step Setting: Evidence from MNIST Auxiliary Logit Distillation Experiment",
  "summary": "In the MNIST auxiliary logit distillation experiment, a student can acquire an unintended teacher trait despite distilling only on no-class logits through a phenomenon called subliminal learning. Under a single-step gradient descent assumption, subliminal learning theory attributes this effect to alignment between the trait and distillation gradients, but does not guarantee that this alignment persists in a multi-step setting. We empirically show that gradient alignment remains weakly but consis",
  "authors": "Chayanon Kitkana, Shivam Arora",
  "category": "research",
  "topics": "safety-alignment,children-education",
  "orgs": null,
  "regions": null,
  "published_at": "2026-04-28T15:46:18.000Z",
  "fetched_at": "2026-07-14T16:31:35.576Z",
  "source_slug": "arxiv-ethics",
  "source_name": "arXiv",
  "source_homepage": "https://arxiv.org",
  "ethics_ai_record_url": "https://ethics.ai/record/5251",
  "original_url": "https://arxiv.org/abs/2604.25779v1",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}