{
  "id": 4204,
  "url": "https://arxiv.org/abs/2605.17246v1",
  "title": "Fidelity Probes for Specification--Code Alignment",
  "summary": "We introduce fidelity probes: natural-language questions generated from a reference artifact with code-derived ground-truth answers, answered from a candidate specification. The fraction of agreeing probes, which we call the fidelity, decomposes into contradiction and coverage-gap rates that drive targeted spec edits to convergence. On a 15-program, roughly 12k-line COBOL benchmark (AWS CardDemo), we raise frozen-test specification fidelity from 0.63 to 0.94 over eight iterations, with the plate",
  "authors": "Ferhat Erata, Hao Zhou, Luke Huan",
  "category": "research",
  "topics": "safety-alignment",
  "orgs": "amazon",
  "regions": null,
  "published_at": "2026-05-17T04:05:54.000Z",
  "fetched_at": "2026-07-14T16:30:50.571Z",
  "source_slug": "arxiv-ethics",
  "source_name": "arXiv",
  "source_homepage": "https://arxiv.org",
  "ethics_ai_record_url": "https://ethics.ai/record/4204",
  "original_url": "https://arxiv.org/abs/2605.17246v1",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}