{
  "id": 4045,
  "url": "https://arxiv.org/abs/2605.24005v2",
  "title": "LC-ERD: Mining Latent Logic for Self-Evolving Reasoning via Consistency-Regulated Reward Decomposition",
  "summary": "The evolution of Large Language Model (LLM) reasoning is bottlenecked by the scarcity of high-quality process data. While self-alignment via endogenous rewards offers a solution, mining valid supervision faces three challenges: (1) Label Noise via Mimetic Bias, where rewards prioritize statistical likelihood over logical truth, creating a \"correctness illusion\" that masks compounding errors; (2) Coarse-Grained Supervision, where sparse global outcomes (e.g., in GRPO) fail to provide granular gui",
  "authors": "Yanyu Chen, Jiyue Jiang, Dianzhi Yu, Zheng Wu, Jiahong Liu, Jiaming Han et al.",
  "category": "research",
  "topics": "bias-fairness,regulation,safety-alignment",
  "orgs": null,
  "regions": null,
  "published_at": "2026-05-19T07:27:50.000Z",
  "fetched_at": "2026-07-14T16:30:41.583Z",
  "source_slug": "arxiv-ethics",
  "source_name": "arXiv",
  "source_homepage": "https://arxiv.org",
  "ethics_ai_record_url": "https://ethics.ai/record/4045",
  "original_url": "https://arxiv.org/abs/2605.24005v2",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}