{
  "id": 16155,
  "url": "https://arxiv.org/abs/2608.01298v1",
  "title": "UDT: Reconciling U-Nets and Diffusion Transformers with Data-Adaptive Token Reduction",
  "summary": "Diffusion Transformers (DiTs) have emerged as a core architecture in generative modeling due to their scalability and adaptability to multimodal tasks. DiTs comprise isotropic transformer blocks, and learn representations progressively across depth, where the denoising objective drives later layers to focus on fine-detail reconstruction. This results in degraded representation quality and an imbalanced encoder-decoder behavior. Prior approaches such as representation alignment (REPA) mitigate th",
  "authors": "Junno Yun, Yaşar Utku Alçalar, Mehmet Akçakaya",
  "category": "research",
  "topics": "safety-alignment",
  "orgs": null,
  "regions": null,
  "published_at": "2026-08-02T15:07:21.000Z",
  "fetched_at": "2026-08-04T05:10:21.797Z",
  "source_slug": "arxiv-cslg",
  "source_name": "arXiv cs.LG",
  "source_homepage": "https://arxiv.org/list/cs.LG/recent",
  "ethics_ai_record_url": "https://ethics.ai/record/16155",
  "original_url": "https://arxiv.org/abs/2608.01298v1",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}