{
  "id": 18694,
  "url": "https://arxiv.org/abs/2608.11034v1",
  "title": "SCOUT: Symmetric Consensus Outlier Detection for Failure Localization in LLM Pre-Training",
  "summary": "In LLM pre-training, synchronization propagates rank-local stalls, slowdowns, and numerical errors into job-wide symptoms, obscuring their origin. Existing diagnosis often relies on in-process monitors that cannot report after the trainer blocks or terminates, or on post-mortem logs that preserve only synchronized symptoms; offline health tests lose the workload and operating conditions that triggered the failure. We present SCOUT, a unified runtime failure-localization framework built on one de",
  "authors": "Zhuang Wang",
  "category": "research",
  "topics": "jobs-economy,healthcare",
  "orgs": null,
  "regions": null,
  "published_at": "2026-08-11T15:12:14.000Z",
  "fetched_at": "2026-08-12T05:10:43.828Z",
  "source_slug": "arxiv-cslg",
  "source_name": "arXiv cs.LG",
  "source_homepage": "https://arxiv.org/list/cs.LG/recent",
  "ethics_ai_record_url": "https://ethics.ai/record/18694",
  "original_url": "https://arxiv.org/abs/2608.11034v1",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}