{
  "id": 4561,
  "url": "https://arxiv.org/abs/2605.18801v1",
  "title": "Position: Let's Develop Data Probes to Fundamentally Understand How Data Affects LLM Performance",
  "summary": "Data is fundamental to large language models (LLMs). However, understanding of what makes certain data useful for different stages of an LLM workflow, including training, tuning, alignment, in-context learning, etc., and why, remains an open question. Current approaches rely heavily on extensive experimentation with large public datasets to obtain empirical heuristics for data filtering and dataset construction. These approaches are compute intensive and lack a principled way of understanding th",
  "authors": "Shiqiang Wang, Herbert Woisetschläger, Hans Arno Jacobsen, Mingyue Ji",
  "category": "research",
  "topics": "safety-alignment",
  "orgs": null,
  "regions": null,
  "published_at": "2026-05-11T11:44:40.000Z",
  "fetched_at": "2026-07-14T16:31:03.582Z",
  "source_slug": "arxiv-ethics",
  "source_name": "arXiv",
  "source_homepage": "https://arxiv.org",
  "ethics_ai_record_url": "https://ethics.ai/record/4561",
  "original_url": "https://arxiv.org/abs/2605.18801v1",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}