{
  "id": 5276,
  "url": "https://arxiv.org/abs/2604.25224v2",
  "title": "ValueBlindBench: Agreement-Gated Stress Testing of LLM-Judged Investment Rationales Before Returns Are Observable",
  "summary": "LLM-based financial agents increasingly produce investment rationales before the outcomes needed to evaluate them are observable. This creates a delayed-ground-truth evaluation problem: realized returns remain the eventual arbiter of investment quality, but they arrive too late and are too noisy to guide many model-development and governance decisions. LLM judges offer a tempting shortcut for pre-deployment evaluation of AI-finance systems, but unvalidated judges may reward verbosity, confidence",
  "authors": "Sidi Chang, Peiying Zhu, Yuxiao Chen",
  "category": "research",
  "topics": "regulation,agents-autonomy,finance-investment",
  "orgs": null,
  "regions": null,
  "published_at": "2026-04-28T05:04:20.000Z",
  "fetched_at": "2026-07-14T16:31:40.218Z",
  "source_slug": "arxiv-ethics",
  "source_name": "arXiv",
  "source_homepage": "https://arxiv.org",
  "ethics_ai_record_url": "https://ethics.ai/record/5276",
  "original_url": "https://arxiv.org/abs/2604.25224v2",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}