{
  "id": 3582,
  "url": "https://arxiv.org/abs/2605.28591v3",
  "title": "Models That Know How Evaluations Are Designed Score Safer",
  "summary": "The validity of AI safety evaluations depends on models behaving consistently across controlled and deployment settings. Prior work has identified test-time contextual cues, such as hypothetical scenarios, as a source of verbalized evaluation awareness and subsequent behavioral shift. In this paper, we investigate a potential explanation of this phenomenon: evaluation meta-knowledge, defined as parametric knowledge about the structural traits that characterize evaluations. Similar to dataset con",
  "authors": "Katharina Deckenbach, Haritz Puerto, Jonas Geiping, Sahar Abdelnabi",
  "category": "research",
  "topics": "safety-alignment,finance-investment",
  "orgs": null,
  "regions": null,
  "published_at": "2026-05-27T15:11:35.000Z",
  "fetched_at": "2026-07-14T16:30:23.244Z",
  "source_slug": "arxiv-ethics",
  "source_name": "arXiv",
  "source_homepage": "https://arxiv.org",
  "ethics_ai_record_url": "https://ethics.ai/record/3582",
  "original_url": "https://arxiv.org/abs/2605.28591v3",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}