{
  "id": 5185,
  "url": "https://arxiv.org/abs/2604.27374v1",
  "title": "Measurement Risk in Supervised Financial NLP: Rubric and Metric Sensitivity on JF-ICR",
  "summary": "As LLMs become credible readers of earnings calls, investor-relations Q\\&A, guidance, and disclosure language, supervised financial NLP benchmarks increasingly function as decision evidence for model selection and deployment. A hidden assumption is that gold labels make such evidence objective. This assumption breaks down when the benchmark ruler itself is sensitive to rubric wording, metric choice, or aggregation policy. We study this measurement risk on Japanese Financial Implicit-Commitment R",
  "authors": "Sidi Chang, Peiying Zhu, Yuxiao Chen, Rongdong Chai",
  "category": "research",
  "topics": "regulation,transparency,finance-investment",
  "orgs": null,
  "regions": "japan",
  "published_at": "2026-04-30T03:39:14.000Z",
  "fetched_at": "2026-07-14T16:31:35.573Z",
  "source_slug": "arxiv-ethics",
  "source_name": "arXiv",
  "source_homepage": "https://arxiv.org",
  "ethics_ai_record_url": "https://ethics.ai/record/5185",
  "original_url": "https://arxiv.org/abs/2604.27374v1",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}