{
  "id": 5961,
  "url": "https://arxiv.org/abs/2604.11070v1",
  "title": "PRISM Risk Signal Framework: Hierarchy-Based Red Lines for AI Behavioral Risk",
  "summary": "Current approaches to AI safety define red lines at the case level: specific prompts, specific outputs, specific harms. This paper argues that red lines can be set more fundamentally -- at the level of value, evidence, and source hierarchies that govern AI reasoning. Using the PRISM (Profile-based Reasoning Integrity Stack Measurement) framework, we define a taxonomy of 27 behavioral risk signals derived from structural anomalies in how AI systems prioritize values (L4), weight evidence types (L",
  "authors": "Seulki Lee",
  "category": "research",
  "topics": "safety-alignment",
  "orgs": null,
  "regions": null,
  "published_at": "2026-04-13T06:50:23.000Z",
  "fetched_at": "2026-07-14T16:32:06.471Z",
  "source_slug": "arxiv-ethics",
  "source_name": "arXiv",
  "source_homepage": "https://arxiv.org",
  "ethics_ai_record_url": "https://ethics.ai/record/5961",
  "original_url": "https://arxiv.org/abs/2604.11070v1",
  "evidence_status": "source-only",
  "attribution": "via ethics.ai"
}